"""CDK stack for the Coupa PO email ingestion pipeline.""" import aws_cdk as cdk from aws_cdk import ( Duration, RemovalPolicy, Stack, aws_cloudwatch as cloudwatch, aws_cloudwatch_actions as cw_actions, aws_dynamodb as dynamodb, aws_kms as kms, aws_lambda as lambda_, aws_lambda_event_sources as lambda_event_sources, aws_s3 as s3, aws_s3_notifications as s3n, aws_ses as ses, aws_ses_actions as ses_actions, aws_secretsmanager as secretsmanager, aws_sns as sns, aws_ssm as ssm, ) from constructs import Construct import common class PoIngestStack(Stack): def __init__(self, scope: Construct, construct_id: str, **kwargs): super().__init__(scope, construct_id, **kwargs) # --- Shared alarm SNS topic (site-alerts) --- # Imported once near the top so every alarm in this stack reuses the same # Topic construct instance (avoids duplicate logical IDs). ALARM-only # SnsAction; no OK action, per the CloudWatch-alarm preference. The # topic's CMK (alias/seahaven-alarm-topics) lives on the topic itself. alarm_topic = sns.Topic.from_topic_arn( self, "SiteAlertsTopic", f"arn:aws:sns:{self.region}:{self.account}:site-alerts", ) # --- S3 bucket for raw emails --- email_bucket = common.make_email_bucket(self, "EmailBucket", "po-ingest-emails") # --- Shared customer-managed CMK for sensitive DynamoDB tables --- # Owned by the account-baseline app (alias/seahaven-dynamodb, INFRA-95 / # M-3); ARN published to SSM. The purchase-orders table was migrated to # SSE-KMS out-of-band, so declaring encryption_key here reconciles the # drift and — via grant_read_write_data below — propagates the required # kms:Decrypt/GenerateDataKey/DescribeKey to the consumer roles. dynamodb_cmk = kms.Key.from_key_arn( self, "DynamoDbCmk", ssm.StringParameter.value_for_string_parameter( self, "/seahaven/dynamodb/cmk-arn" ), ) # --- Purchase-orders DynamoDB table --- # Owned by this stack. Streams enabled for the site-extractor pipeline. # Other stacks (seahaven-slack-bot) reference this table via fromTableName(). po_table = dynamodb.Table( self, "PurchaseOrdersTable", table_name="purchase-orders", partition_key=dynamodb.Attribute( name="po_number", type=dynamodb.AttributeType.STRING, ), billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST, removal_policy=RemovalPolicy.RETAIN, stream=dynamodb.StreamViewType.NEW_IMAGE, encryption=dynamodb.TableEncryption.CUSTOMER_MANAGED, encryption_key=dynamodb_cmk, ) # --- Anthropic API key secret removed (Bedrock migration) --- # PO parsing stays fully AI but moved from the Anthropic API to the # Bedrock inference profile us.anthropic.claude-haiku-4-5-20251001-v1:0, # so no provider API key is needed. The old secret # "po-ingest/anthropic-api-key" had RemovalPolicy.RETAIN, so it is # ORPHANED (not deleted) by this change: delete it manually post-deploy # and revoke the stored key at Anthropic. # --- DLQ for failed async invocations (INFRA-41 / audit H-8) --- # SES → S3 → Lambda is async; without an OnFailure destination a failed # parse (bad email, transient error) is silently dropped after Lambda's # retries. CDK generates the queue name to avoid colliding with the # interim CLI-created po-email-processor-dlq (removed post-deploy). email_processor_dlq = common.make_processor_dlq(self, "EmailProcessorDlq") # --- Lambda function --- email_processor_log_group = common.make_function_log_group( self, "EmailProcessor", "po-email-processor" ) email_processor = lambda_.Function( self, "EmailProcessor", function_name="po-email-processor", runtime=lambda_.Runtime.PYTHON_3_12, architecture=lambda_.Architecture.ARM_64, handler="handler.handler", code=lambda_.Code.from_asset( "../lambdas", exclude=["**/__pycache__/**", "**/tests/**", "**/package/**"], bundling=cdk.BundlingOptions( image=lambda_.Runtime.PYTHON_3_12.bundling_image, command=[ "bash", "-c", # NOTE: non-recursive glob (not `cp -r`) so tests/ and # the stale package/ dir are never shipped -- only # top-level .py siblings of handler.py. This replaces # a hand-maintained four-file allowlist that twice # nearly shipped a broken Lambda (missing # template_parser in PR #105, nearly missing # derived_fields in PR #2) because a new sibling # import wasn't added to the list. The glob makes # that class of bug structurally impossible; # tests/test_bundle_consistency.py ast-parses # handler.py's first-party imports and asserts this # command ships all of them, so a future revert back # to an allowlist that omits a sibling fails CI. # pip step removed in Phase 7: requirements.txt is now # empty (boto3 comes from the Lambda runtime), so nothing # is installed and the manylinux pin has nothing to pin. # shared/*.py ships the four modules extracted to # lambdas/shared/ (Phase 3): ses_auth, web_ui_auth, # email_parsing, emf. Flat cp keeps the bare-name # imports (e.g. `from ses_auth import ...`) resolving # unchanged in /asset-output. "cp po/email_processor/*.py /asset-output/ && " "cp shared/*.py /asset-output/", ], ), ), timeout=Duration.seconds(60), memory_size=256, log_group=email_processor_log_group, dead_letter_queue=email_processor_dlq, environment={ "PO_TABLE": "purchase-orders", "BEDROCK_MODEL_ID": "us.anthropic.claude-haiku-4-5-20251001-v1:0", # Fail-closed sender auth (INFRA-107): the handler only # accepts mail whose SES-stamped Authentication-Results # header carries dkim=pass for one of these domains. # Observed on live traffic 2026-07-15: Coupa PO mail passes # DKIM for amazon.coupahost.com (and amazonses.com, which is # deliberately NOT allowlisted — every SES customer's mail # passes that). Unset/empty ⇒ the handler rejects all mail. "ALLOWED_DKIM_DOMAINS": "amazon.coupahost.com", }, ) # Grant permissions email_bucket.grant_read(email_processor) po_table.grant_read_write_data(email_processor) # --- Bedrock InvokeModel grant --- # The us.* inference profile can route cross-region, so the grant MUST # cover both the inference-profile ARN AND the per-region foundation-model # ARNs (empty account field) for every region the profile can reach # (us-east-1/us-east-2/us-west-2). A profile-only grant AccessDenies at # runtime whenever the profile routes to a region whose foundation-model # ARN is not allowed. email_processor.add_to_role_policy(common.make_bedrock_invoke_statement(self)) # --- Standard per-Lambda alarms: po-email-processor --- # errors (INFRA-41 / audit H-8), throttles, DLQ-messages (dropped PO # emails), and a p99 duration alarm (orphan adoption of the CLI # Lambda-Duration-po-email-processor under -duration naming, 45000 ms # = 75% of the 60s timeout, eval 3 / dp 2). All ALARM-only to site-alerts. common.add_standard_lambda_alarms( self, "EmailProcessor", email_processor, "po-email-processor", alarm_topic, duration_statistic="p99", errors=True, dlq=email_processor_dlq, descriptions={ "errors": "po-email-processor async invocation errors", "throttles": "po-email-processor invocation throttles", "dlq": "po-email-processor DLQ has messages (dropped PO emails)", "duration": "po-email-processor p99 duration approaching the 60s timeout", }, ) # --- Sender-auth rejection alarm (INFRA-107) --- # A rejected email (bad/unaligned DKIM verdict) returns normally, so it # produces NO Lambda error, NO DLQ message and NO retry -- only a # `sender_auth_rejected` warning log. Without this metric filter + alarm a # domain drift (Coupa rotates its signing subdomain, SES changes its # Authentication-Results format, the allowlist is wrong) would silently # discard 100% of legitimate PO mail while every other alarm stays green. # A CloudWatch Logs metric filter turns those warnings into a metric so a # false-reject storm pages instead of vanishing. default_value=0 keeps the # series populated (alarm stays OK, never INSUFFICIENT_DATA) between events. common.add_sender_auth_rejected_alarm( self, "EmailProcessor", "po-email-processor", alarm_topic, email_processor_log_group, ) # --- Template fallback-rate alarm: po-email-processor --- # The processor tries a deterministic template parse first and only calls # the Bedrock AI extractor on a miss/invalid. A sustained rise in the # ai_fallback share signals Coupa template drift (coverage collapse). # EMF metric Seahaven/PoIngest/ParseOutcome, dimensioned by ParseMethod # (template|ai_fallback). # # RETUNED for PO volume (~57 emails/day ≈ 14.25 per 6h period) -- the WO # alarm's 15-min period / >=10-sample floor assume ~760/day and would be # structurally DEAD here (a 15-min period holds ~0.6 PO emails, so the # floor is never met and the IF always takes the 0 branch): # * period 6h: a stable ~14-email denominator per datapoint. # * volume floor >=8: at the floor, one fallback email = 12.5% < 20%, # so a single email can NEVER breach a datapoint; a breach needs >=2 # fallbacks in one 6h window (2/8 = 25%) or >=3 at typical volume # (3/14 ≈ 21%). Sparse overnight/weekend windows (<8 emails) take # the 0 branch -- non-breaching by design (accepted trade: a Friday- # evening drift may not page until weekend volume accrues). # * threshold >20%: expected baseline fallback ≈1% (comments 0.55% + # multi-line 0.18% + non-USD 0) -- far below the threshold. # * 2 of 4 datapoints (24h span): isolated noise self-clears, while # total template drift (100% fallback) pages within ~12h. # Post-#102 rule: NO element-wise MAX(timeseries, scalar) in alarm math; # the IF volume floor guarantees the non-zero denominator. Any change to # this expression must be gated by `npx cdk synth po-ingest`. # # DOUBLE-COUNT ACCOUNTING (Phase 1 / PO AI-fallback gate): PO emits # ParseMethod=ai_fallback BEFORE the Bedrock call for EVERY AI-path # email (handler pre-call emit; try_deterministic_parse returns # "ai_fallback" on every template miss), so a gate-rejected email # already appears exactly once in `fb`. Therefore fb = ALL fallback # attempts (accepted + rejected), fb + tmpl = ALL emails, and # rate = fb/(fb+tmpl) is exact -- the expression below is deliberately # left BYTE-IDENTICAL to the pre-Phase-1 form, and `rej` # (ai_fallback_rejected) is deliberately EXCLUDED from this # expression's numerator, denominator, and volume floor, and is never # added to using_metrics. This is NOT an oversight: folding `rej` in # here as WO does (fb+rej numerator / fb+rej+tmpl denominator) would # double-count every rejected email in both numerator and # denominator (PO's pre-call emit already counts it once via `fb`), # inflating the observed rate toward 100% and double-counting toward # the >=8 volume floor -- a prompt-injection probing burst would then # falsely page this template-drift alarm on top of the dedicated # rejected alarm below. The rejected series gets its own alarm # instead (EmailProcessorAiFallbackRejectedAlarm, below). common.make_fallback_rate_alarm( self, "EmailProcessorTemplateFallbackRateAlarm", namespace="Seahaven/PoIngest", alarm_topic=alarm_topic, alarm_name="po-email-processor-template-fallback-rate", alarm_description=( "po-email-processor deterministic-template coverage collapse: " ">20% of parses fell back to the Bedrock AI extractor" ), rejected_included=False, period=Duration.hours(6), threshold=20, floor=8, evaluation_periods=4, datapoints_to_alarm=2, ) # --- AI-fallback rejected alarm: po-email-processor (Phase 1) --- # The validate_ai_fallback gate (template_parser.py) fail-closes Bedrock # output that doesn't match PO's contract (structurally wrong shape, # injected po_number/email_type, wrong field types) and emits # ParseMethod=ai_fallback_rejected instead of writing it. That is a # SILENT skip (`continue`, never raise) by design -- attacker-controlled # input must not churn the retry/DLQ path -- so without a dedicated # alarm a sustained rejection run (prompt-injection probing, or a # template-drift outage whose AI output also happens to fail the gate) # is invisible everywhere except this metric and the ReasonCode log # line. # # RETUNED for PO volume (~57 emails/day, baseline ai_fallback rate # ~1% => ~0.6 AI-fallback emails/day, expected rejections ~= 0) -- NOT # WO's 5-min/2-of-6 sparse idiom (wo_stack.py), which needs two # rejections inside a single 30-min window and is structurally dead at # this volume. Mirrors the PO fallback-rate alarm's 6h/eval-4/dp-2 # retune idiom above, but with a COUNT floor on the rejected series # itself rather than an email-volume floor: an email-volume floor # (fb+tmpl>=N) would suppress paging in exactly the sparse # overnight/weekend windows where a silently-dropped email matters # most, and there is no denominator here, so there is nothing else to # guard against divide-by-zero. FILL(rej,0) turns the sparse EMF # series (no datapoint in quiet periods -- no metric-filter # default_value exists for EMF) into a dense 0-series so every # evaluation window has data. Post-#102 rule still holds: NO # element-wise MAX(timeseries, scalar) anywhere in this expression. # # Tuning: rejections self-clear unless >=2 breaching datapoints land in # >=2 distinct 6h windows within 24h (sustained probing, or template # drift whose AI output also fails the gate), which pages within # ~12-24h. Accepted residual (matches WO's accepted residual): because # the breach is measured per 6h window, ANY burst of rejections # confined to a single 6h window -- whether one stray email or dozens # in a 20-minute spike -- is one breaching datapoint and never pages # this alarm by itself. This is deliberate anti-flap tuning at ~0 # expected rejections/day, not a coverage gap in the fail-closed gate: # every burst email is still rejected before any DynamoDB write, and # the burst stays fully visible as ai_fallback_rejected datapoints and # ReasonCode log lines, with the pre-call ai_fallback emit also raising # the fallback-rate numerator above. A same-window burst detector # (1-of-1 at a higher threshold) is a tracked follow-up if faster # single-window paging is wanted. rejected_metric = cloudwatch.Metric( namespace="Seahaven/PoIngest", metric_name="ParseOutcome", dimensions_map={"ParseMethod": "ai_fallback_rejected"}, statistic="Sum", period=Duration.hours(6), ) rejected_floor = cloudwatch.MathExpression( expression="IF(FILL(rej,0)>=1, FILL(rej,0), 0)", using_metrics={"rej": rejected_metric}, period=Duration.hours(6), label="AiFallbackRejectedCount", ) rejected_floor.create_alarm( self, "EmailProcessorAiFallbackRejectedAlarm", alarm_name="po-email-processor-ai-fallback-rejected", alarm_description=( "po-email-processor is rejecting Bedrock AI-fallback output at " "the validation gate (possible prompt-injection probing or " "template drift silently dropping real mail)" ), threshold=1, comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, evaluation_periods=4, datapoints_to_alarm=2, treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) # S3 event notification → Lambda email_bucket.add_event_notification( s3.EventType.OBJECT_CREATED, s3n.LambdaDestination(email_processor), s3.NotificationKeyFilter(prefix="inbound/"), ) # --- SES Receipt Rule --- # Reuse the existing INBOUND_MAIL rule set (shared with workorder-ingest) rule_set = ses.ReceiptRuleSet.from_receipt_rule_set_name( self, "ExistingRuleSet", "INBOUND_MAIL", ) rule_set.add_rule( "PoEmailRule", recipients=["amazon_po@int.seahaven.com"], actions=[ ses_actions.S3( bucket=email_bucket, object_key_prefix="inbound/", ), ], ) # --- Web UI auth token secret --- # Shared secret for the web UI auth gate, stored in Secrets Manager and # resolved at runtime so the token never appears in CloudFormation templates # or Lambda environment variables. Create this secret before deploying # either stack; both PO and WO stacks reference it by name. web_ui_auth_secret = secretsmanager.Secret.from_secret_name_v2( self, "WebUiAuthToken", "procurement-ingest/web-ui-auth-token", ) # --- Web UI Lambda --- web_ui_log_group = common.make_function_log_group(self, "WebUI", "po-web-ui") web_ui = lambda_.Function( self, "WebUI", function_name="po-web-ui", runtime=lambda_.Runtime.PYTHON_3_12, architecture=lambda_.Architecture.ARM_64, handler="handler.handler", code=lambda_.Code.from_asset( "../lambdas", exclude=["**/__pycache__/**", "requirements.txt"], bundling=cdk.BundlingOptions( image=lambda_.Runtime.PYTHON_3_12.bundling_image, command=[ "bash", "-c", # web_ui_auth.py is shared (lambdas/shared) and must land # FLAT beside handler.py so the bare # `from web_ui_auth import is_authenticated` resolves at # runtime. Only web_ui_auth is copied from shared/ -- the # other shared modules (ses_auth/email_parsing/emf) are # email-processor-only and must not bloat the web UI zip. # requirements.txt stays excluded (Dependabot anchor only, # never runtime): keeps the deployed file list = {handler, # web_ui_auth}. NOTE: the top-level `exclude=` on # from_asset only filters the asset-hash fingerprint, NOT # the directory Docker bundling actually mounts, so a # local __pycache__/requirements.txt on disk at synth # time WOULD otherwise leak into the bundled zip -- strip # them explicitly post-cp instead of relying on exclude. "cp -r po/web_ui/. /asset-output/ && " "cp shared/web_ui_auth.py /asset-output/ && " "rm -rf /asset-output/__pycache__ /asset-output/requirements.txt", ], ), ), timeout=Duration.seconds(60), memory_size=256, log_group=web_ui_log_group, environment={ "PO_TABLE": "purchase-orders", # Defense-in-depth shared secret for the web UI handler. The # handler fails closed if this ARN is unset or the secret is # missing, so any future invocation path cannot re-expose the # PO DB unauthenticated. The secret value is fetched at runtime # from Secrets Manager (not embedded in env vars or template). "WEB_UI_AUTH_TOKEN_SECRET_ARN": web_ui_auth_secret.secret_arn, }, ) po_table.grant_read_data(web_ui) web_ui_auth_secret.grant_read(web_ui) # --- Standard per-Lambda alarms: po-web-ui --- # Throttles + p99 duration only (no errors alarm, no DLQ -- web_ui is a # synchronous read path with no async DLQ). p99 / 45000 ms (75% of the # 60s timeout) / eval 3, datapoints 2. common.add_standard_lambda_alarms( self, "WebUi", web_ui, "po-web-ui", alarm_topic, duration_statistic="p99", errors=False, dlq=None, descriptions={ "throttles": "po-web-ui invocation throttles", "duration": "po-web-ui p99 duration approaching the 60s timeout", }, ) # Public Function URL removed 2026-06-08 (INFRA-74 / audit C-5): the # unauthenticated FunctionUrlAuthType.NONE URL was deleted out-of-band # via CLI. Removing the construct (and its auto-generated Principal:* # invoke permission) reconciles IaC with the live state. # --- Verified sites table (extracted from PO ship-to addresses) --- verified_sites_table = dynamodb.Table( self, "VerifiedSitesTable", table_name="verified-sites", partition_key=dynamodb.Attribute( name="siteCode", type=dynamodb.AttributeType.STRING, ), billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST, removal_policy=RemovalPolicy.RETAIN, ) # by-state GSI removed 2026-06-03 (audit M-20): 0 reads in 30d against # 518 WCU of write amplification. Re-add if a state-level query path ships. # --- Site extractor Lambda (DynamoDB Streams → verified-sites) --- site_extractor_log_group = common.make_function_log_group( self, "SiteExtractor", "po-ingest-site-extractor" ) site_extractor = lambda_.Function( self, "SiteExtractor", function_name="po-ingest-site-extractor", runtime=lambda_.Runtime.PYTHON_3_12, architecture=lambda_.Architecture.ARM_64, handler="handler.handler", code=lambda_.Code.from_asset( # requirements.txt is excluded from the bundle: it exists only # as a Dependabot anchor (git-based scan sees it), never pip- # installed (this is a plain non-bundled asset) and never needed # at runtime (boto3 comes from the Lambda runtime). Excluding it # keeps the deployed asset hash neutral vs base while the manifest # still lands in git for Dependabot. "../lambdas/po/site_extractor", exclude=["**/__pycache__/**", "requirements.txt"], ), timeout=Duration.seconds(60), memory_size=256, log_group=site_extractor_log_group, environment={ "VERIFIED_SITES_TABLE": verified_sites_table.table_name, "PENDING_REVIEW_TABLE": "pending-site-review", }, ) verified_sites_table.grant_read_write_data(site_extractor) site_extractor.add_event_source( lambda_event_sources.DynamoEventSource( po_table, starting_position=lambda_.StartingPosition.TRIM_HORIZON, batch_size=10, max_batching_window=Duration.seconds(30), bisect_batch_on_error=True, retry_attempts=3, ) ) # --- Standard per-Lambda alarms: po-ingest-site-extractor --- # errors + throttles + p99 duration. NO DLQ alarm: site_extractor is a # DynamoEventSource stream consumer with no async DLQ attached (dlq=None). # Stream-consumer errors retry per the event-source config, but a # persistent failure stalls the verified-sites pipeline. p99 / 45000 ms # (75% of the 60s timeout) / eval 3, datapoints 2. common.add_standard_lambda_alarms( self, "SiteExtractor", site_extractor, "po-ingest-site-extractor", alarm_topic, duration_statistic="p99", errors=True, dlq=None, descriptions={ "errors": "po-ingest-site-extractor invocation errors", "throttles": "po-ingest-site-extractor invocation throttles", "duration": "po-ingest-site-extractor p99 duration approaching the 60s timeout", }, ) cdk.CfnOutput( self, "VerifiedSitesTableName", value=verified_sites_table.table_name, description="Verified site addresses extracted from POs", ) # --- Pending site review table (POs with no extractable site code) --- pending_review_table = dynamodb.Table( self, "PendingSiteReviewTable", table_name="pending-site-review", partition_key=dynamodb.Attribute( name="po_number", type=dynamodb.AttributeType.STRING, ), billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST, removal_policy=RemovalPolicy.RETAIN, ) pending_review_table.grant_read_write_data(site_extractor) verified_sites_table.grant_read_data(site_extractor) # --- DynamoDB throttle + system-error alarms --- # ThrottledRequests / SystemErrors emit at TableName + Operation only # (verified against live CloudWatch: no TableName-only rollup exists, and # metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.261.0). # Each table currently has zero throttle/error datapoints, so the series # only materialise on first occurrence — NOT_BREACHING keeps them OK until # then. common.add_ddb_alarms( self, "PurchaseOrdersTable", po_table, "purchase-orders", alarm_topic ) common.add_ddb_alarms( self, "VerifiedSitesTable", verified_sites_table, "verified-sites", alarm_topic, ) common.add_ddb_alarms( self, "PendingSiteReviewTable", pending_review_table, "pending-site-review", alarm_topic, ) # --- Function ARN + consumed-table-name outputs (Phase 4, additive) --- cdk.CfnOutput( self, "EmailProcessorFunctionArn", value=email_processor.function_arn, description="ARN of the po-email-processor Lambda", ) cdk.CfnOutput( self, "WebUiFunctionArn", value=web_ui.function_arn, description="ARN of the po-web-ui Lambda", ) cdk.CfnOutput( self, "SiteExtractorFunctionArn", value=site_extractor.function_arn, description="ARN of the po-ingest-site-extractor Lambda", ) cdk.CfnOutput( self, "PurchaseOrdersTableName", value=po_table.table_name, description="purchase-orders DynamoDB table consumed by this stack", ) cdk.CfnOutput( self, "PendingSiteReviewTableName", value=pending_review_table.table_name, description="pending-site-review DynamoDB table", )