2026-04-07 12:12:30 -04:00
|
|
|
"""CDK stack for the Coupa PO email ingestion pipeline."""
|
|
|
|
|
|
|
|
|
|
import aws_cdk as cdk
|
|
|
|
|
from aws_cdk import (
|
|
|
|
|
Duration,
|
|
|
|
|
RemovalPolicy,
|
|
|
|
|
Stack,
|
Reconcile IaC with out-of-band DLQ + Function URL changes (INFRA-74, INFRA-41) (#50)
Make CDK the source of truth for two sets of changes applied out-of-band
via CLI to the po-ingest and WorkorderIngestStack stacks.
INFRA-74 (audit C-5): remove the public FunctionUrlAuthType.NONE Function
URL construct (and its auto-generated Principal:* invoke permission +
output) from both po-web-ui and workorder-web-ui. The URLs were already
deleted live via CLI; CFN's delete is idempotent.
INFRA-41 (audit H-8): add a CDK-managed SQS dead-letter queue
(dead_letter_queue=, 14d retention, SSL-enforced, CDK-generated name) and
an ALARM-only Errors alarm (Sum, threshold>0, site-alerts topic) for both
po-email-processor and workorder-email-processor, mirroring the
apm-wo-analysis-classifier DLQ and payments-payroll-batch alarm patterns.
Interim CLI resources (per-fn -dlq queues, -errors alarms, dlq-send inline
policies, OnFailure event-invoke-configs) removed post-deploy.
2026-06-08 16:02:29 -04:00
|
|
|
aws_cloudwatch as cloudwatch,
|
|
|
|
|
aws_cloudwatch_actions as cw_actions,
|
2026-04-07 12:12:30 -04:00
|
|
|
aws_dynamodb as dynamodb,
|
2026-06-10 19:31:55 -04:00
|
|
|
aws_kms as kms,
|
2026-04-07 12:12:30 -04:00
|
|
|
aws_lambda as lambda_,
|
2026-04-30 14:26:53 -04:00
|
|
|
aws_lambda_event_sources as lambda_event_sources,
|
|
|
|
|
aws_logs as logs,
|
2026-04-07 12:12:30 -04:00
|
|
|
aws_s3 as s3,
|
|
|
|
|
aws_s3_notifications as s3n,
|
|
|
|
|
aws_ses as ses,
|
|
|
|
|
aws_ses_actions as ses_actions,
|
|
|
|
|
aws_secretsmanager as secretsmanager,
|
Reconcile IaC with out-of-band DLQ + Function URL changes (INFRA-74, INFRA-41) (#50)
Make CDK the source of truth for two sets of changes applied out-of-band
via CLI to the po-ingest and WorkorderIngestStack stacks.
INFRA-74 (audit C-5): remove the public FunctionUrlAuthType.NONE Function
URL construct (and its auto-generated Principal:* invoke permission +
output) from both po-web-ui and workorder-web-ui. The URLs were already
deleted live via CLI; CFN's delete is idempotent.
INFRA-41 (audit H-8): add a CDK-managed SQS dead-letter queue
(dead_letter_queue=, 14d retention, SSL-enforced, CDK-generated name) and
an ALARM-only Errors alarm (Sum, threshold>0, site-alerts topic) for both
po-email-processor and workorder-email-processor, mirroring the
apm-wo-analysis-classifier DLQ and payments-payroll-batch alarm patterns.
Interim CLI resources (per-fn -dlq queues, -errors alarms, dlq-send inline
policies, OnFailure event-invoke-configs) removed post-deploy.
2026-06-08 16:02:29 -04:00
|
|
|
aws_sns as sns,
|
|
|
|
|
aws_sqs as sqs,
|
2026-06-10 19:31:55 -04:00
|
|
|
aws_ssm as ssm,
|
2026-04-07 12:12:30 -04:00
|
|
|
)
|
|
|
|
|
from constructs import Construct
|
|
|
|
|
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
# Operations these tables actually issue (PutItem/UpdateItem/DeleteItem writes,
|
|
|
|
|
# GetItem/Query/BatchGetItem reads). DynamoDB emits ThrottledRequests/SystemErrors
|
|
|
|
|
# keyed by TableName + Operation only, so the CDK *_for_operations helpers (which
|
|
|
|
|
# render a SUM MathExpression across these per-operation metrics) are the correct,
|
|
|
|
|
# non-deprecated way to roll a table up to a single alarmable series.
|
|
|
|
|
_DDB_ALARM_OPERATIONS = [
|
|
|
|
|
dynamodb.Operation.GET_ITEM,
|
|
|
|
|
dynamodb.Operation.BATCH_GET_ITEM,
|
|
|
|
|
dynamodb.Operation.QUERY,
|
|
|
|
|
dynamodb.Operation.SCAN,
|
|
|
|
|
dynamodb.Operation.PUT_ITEM,
|
|
|
|
|
dynamodb.Operation.UPDATE_ITEM,
|
|
|
|
|
dynamodb.Operation.DELETE_ITEM,
|
|
|
|
|
dynamodb.Operation.BATCH_WRITE_ITEM,
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _add_ddb_alarms(scope, id_prefix, table, alarm_name_prefix, alarm_topic):
|
|
|
|
|
"""Add throttle + system-error alarms for a DynamoDB table.
|
|
|
|
|
|
|
|
|
|
Both fire on any non-zero datapoint in a 5-min window. ALARM-only SnsAction
|
|
|
|
|
to site-alerts (no OK action); TreatMissingData NOT_BREACHING.
|
|
|
|
|
"""
|
|
|
|
|
table.metric_throttled_requests_for_operations(
|
|
|
|
|
operations=_DDB_ALARM_OPERATIONS,
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
scope,
|
|
|
|
|
f"{id_prefix}ThrottlesAlarm",
|
|
|
|
|
alarm_name=f"{alarm_name_prefix}-throttles",
|
|
|
|
|
alarm_description=f"{alarm_name_prefix} DynamoDB throttled requests",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
|
|
|
|
table.metric_system_errors_for_operations(
|
|
|
|
|
operations=_DDB_ALARM_OPERATIONS,
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
scope,
|
|
|
|
|
f"{id_prefix}SystemErrorsAlarm",
|
|
|
|
|
alarm_name=f"{alarm_name_prefix}-system-errors",
|
|
|
|
|
alarm_description=f"{alarm_name_prefix} DynamoDB server-side (5xx) errors",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
2026-04-07 12:12:30 -04:00
|
|
|
|
Add fail-closed SES sender authentication (INFRA-107) (#98)
* Add fail-closed SES sender authentication
The From header and any raw-MIME Authentication-Results copies are
attacker-forgeable, so a forged email to apm@int.seahaven.com or
amazon_po@int.seahaven.com could create or mutate a WO/PO (INFRA-107,
CRITICAL). Both S3-triggered email processors now authenticate the
sender against the Authentication-Results header SES itself prepends
at delivery: only the topmost header is consulted, its authserv-id
must be amazonses.com, and it must carry dkim=pass for a domain in
the per-pipeline ALLOWED_DKIM_DOMAINS env var (comma-separated, set
in CDK so ops can adjust without code changes).
Allowlists come from live traffic observed 2026-07-15 on both ingest
buckets: WO mail arrives via the apm@ Google Groups forward, which
re-signs as seahaven.com (the hxgnsmartcloud.com signature does not
survive the forward); PO mail passes for amazon.coupahost.com.
amazonses.com also passes on PO mail but is deliberately excluded --
every SES customer's outbound mail passes for it.
Every failure path (env var unset, header missing or unparseable,
verdict fail, unaligned domain) rejects the email: a structured
warning with the reason and S3 key is logged and the record skipped
without erroring the invocation, so rejected mail causes no Lambda
retries or DLQ messages. Handler signatures and event sources are
unchanged.
Refs: INFRA-107
* Harden AR parser per cross-family review
Cross-family (GPT-4.1) review findings: terminate the dkim result
token at end-of-clause, whitespace, or a comment so a value like
"dkim=pass-fake" can never be read as a pass; normalize trailing
dots off allowlist entries so "seahaven.com." matches; make the
compat32 parser policy explicit. Adds tests for result-token
boundaries, comments after the result, quoted domain values, and
folding inside a dkim clause.
Refs: INFRA-107
* Harden AR parsing and alarm on sender-auth rejects
The SES-stamped Authentication-Results value echoes attacker-controlled
SMTP-session tokens (envelope-from, helo, header.from) as their own
semicolon-delimited property clauses. A naive split(";") tore an RFC 5321
quoted-local-part MAIL FROM apart and manufactured a forged dkim=pass
clause, so a fully spoofed email was accepted on the genuinely
SES-stamped topmost header. Tokenise comment- and quoted-string-aware
(RFC 8601 / RFC 5322): strip CFWS comments, split clauses only on
semicolons outside a quoted-string, and fail closed on unbalanced
quotes/comments so a ';' inside a quoted pvalue can never start a clause.
Rejected mail returns normally (no error, no retry, no DLQ message), so a
signing-domain drift or a wrong allowlist would silently discard 100% of
legitimate mail while every alarm stayed green. Add a CloudWatch Logs
metric filter + alarm on the sender_auth_rejected warning to both stacks
so a false-reject storm pages instead of vanishing. This is also the
safety net for the WO seahaven.com allowlist assumption, which must be
validated against a live SES-stamped header (a plain Gmail auto-forward
re-signs under the sending Workspace domain, not seahaven.com).
Refs: INFRA-107
* chore: retrigger CI (no run recorded for 7c74ac1)
* Fix quoted-AUID DKIM domain spoof in sender auth
Resolve three confirmed /sh-security-review findings on the fail-closed
SES sender-authentication control.
HIGH: header.i/header.d domain extraction was not quoted-string aware.
An attacker with a valid DKIM key for their own domain could set an
RFC 6376-legal AUID such as i="@seahaven.com"@attacker.com; the naive
extractor stopped at the closing quote and returned seahaven.com,
accepting forged mail. Extraction now tokenises the clause with the same
quoted-string discipline already used for clause splitting: header.d
(the plain signing domain) is authoritative when present, otherwise the
header.i domain is the part after the AUID's LAST top-level "@", so a "@"
inside a quoted local-part is treated as signer-controlled label text and
yields the true signer (attacker.com), not seahaven.com.
LOW: the topmost-header parse ran outside evaluate_sender_authentication's
try/except, so an unexpected parser exception on crafted input could
propagate into the handler and Lambda async retries/DLQ. The parse now
fails CLOSED with an authentication_results_unparseable reason.
MEDIUM: the sender_auth_rejected alarm used Sum>=3 over 15 min, blind to
a low-volume total-reject outage (a trickle that never sums to 3). Both
stacks now alarm on >=1 reject per 5-min period with evaluation_periods=3
/ datapoints_to_alarm=2, so a sustained reject condition pages even at one
reject per period while a lone stray probe self-clears.
Refs: INFRA-107
* Load Lambda function dir on sys.path in tests
Rebasing INFRA-107 onto main folded #95's pytest suite into this
branch's tests. The unified conftest loads the PO/WO handlers by file
path, and handler.py now does `from ses_auth import
authenticate_inbound_email` -- a bare sibling import that resolves in
the Lambda only because the runtime puts each function's own directory
on sys.path. The shared load_handler now adds that directory so the
handler tests import correctly alongside the sender-auth tests.
Refs: INFRA-107
* Note #97 test files in README directory tree
The rebase onto main brought in #97's tests/requirements.txt and
tests/test_po_merge.py. List both in the directory tree so it matches
the tree on disk.
Refs: INFRA-107
* Document INFRA-107 forwarder-binding risk acceptance
Record the accepted risk that WO sender auth binds to the apm@ forward's
re-signing domain (seahaven.com) rather than the Hexagon originator; the
apm@ Google Group's restricted posting policy is the load-bearing control
(escalates to HIGH if the group is opened to external posting). Also
correct the sender-auth-rejected alarm docs to match the shipped config
(>=1 per 5-min, 2-of-3 datapoints, not the superseded >=3/15min) and
note the SES-AR-01/02 parser hardening follow-ups.
Refs: INFRA-107
2026-07-15 20:58:47 -04:00
|
|
|
# CloudWatch namespace for the log-derived sender-authentication metrics.
|
|
|
|
|
_SENDER_AUTH_METRIC_NAMESPACE = "Seahaven/ProcurementIngest"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _add_sender_auth_rejected_alarm(scope, id_prefix, function_name, alarm_topic):
|
|
|
|
|
"""Metric-filter + alarm on ``sender_auth_rejected`` warnings (INFRA-107).
|
|
|
|
|
|
|
|
|
|
A rejected inbound email is skipped without erroring the invocation, so it
|
|
|
|
|
is invisible to the Errors/Throttles/DLQ alarms. This turns the structured
|
|
|
|
|
warning log into a CloudWatch metric and pages when rejections spike --
|
|
|
|
|
catching a silent false-reject storm (allowlist wrong, signing-domain
|
|
|
|
|
drift, SES header-format change) that would otherwise discard legitimate
|
|
|
|
|
mail while the pipeline reports healthy.
|
|
|
|
|
|
|
|
|
|
ALARM-only SnsAction to site-alerts; no OK action. The metric filter reads
|
|
|
|
|
the function's own log group (imported by the deterministic
|
|
|
|
|
``/aws/lambda/<fn>`` name, created by the function's log_retention). A plain
|
|
|
|
|
substring pattern is used because Lambda prefixes each line with its own
|
|
|
|
|
level/timestamp/request-id, so the JSON payload is not a standalone JSON
|
|
|
|
|
log event a `{$.event=...}` pattern could match.
|
|
|
|
|
"""
|
|
|
|
|
metric_name = f"{function_name}-sender-auth-rejected"
|
|
|
|
|
logs.MetricFilter(
|
|
|
|
|
scope,
|
|
|
|
|
f"{id_prefix}SenderAuthRejectedFilter",
|
|
|
|
|
log_group=logs.LogGroup.from_log_group_name(
|
|
|
|
|
scope,
|
|
|
|
|
f"{id_prefix}LogGroup",
|
|
|
|
|
f"/aws/lambda/{function_name}",
|
|
|
|
|
),
|
|
|
|
|
filter_pattern=logs.FilterPattern.literal('"sender_auth_rejected"'),
|
|
|
|
|
metric_namespace=_SENDER_AUTH_METRIC_NAMESPACE,
|
|
|
|
|
metric_name=metric_name,
|
|
|
|
|
metric_value="1",
|
|
|
|
|
default_value=0,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
# Fire on a *sustained* reject condition rather than a volume spike. The
|
|
|
|
|
# earlier Sum>=3-over-15-min threshold had a blind spot that is exactly the
|
|
|
|
|
# failure this alarm exists to catch: a low-traffic pipeline in total
|
|
|
|
|
# drift outage (allowlist wrong / signing-domain changed) may only produce
|
|
|
|
|
# a trickle of rejects -- one every few minutes -- that never sums to 3 in
|
|
|
|
|
# any window, so the outage never pages. Instead: >=1 reject per 5-min
|
|
|
|
|
# period, alarming when 2 of the last 3 periods breach (evaluation_periods=3
|
|
|
|
|
# / datapoints_to_alarm=2, the same idiom as the duration alarm). A single
|
|
|
|
|
# stray spoof probe (one lone period) is tolerated and self-clears, but a
|
|
|
|
|
# sustained reject condition trips within ~10-15 min even at one reject per
|
|
|
|
|
# period. default_value=0 on the metric filter keeps the series continuous
|
|
|
|
|
# so NOT_BREACHING only applies before the first datapoint ever arrives.
|
|
|
|
|
cloudwatch.Metric(
|
|
|
|
|
namespace=_SENDER_AUTH_METRIC_NAMESPACE,
|
|
|
|
|
metric_name=metric_name,
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
scope,
|
|
|
|
|
f"{id_prefix}SenderAuthRejectedAlarm",
|
|
|
|
|
alarm_name=f"{function_name}-sender-auth-rejected",
|
|
|
|
|
alarm_description=(
|
|
|
|
|
f"{function_name} rejected inbound mail on sender authentication "
|
|
|
|
|
"(possible allowlist/DKIM-domain drift silently dropping real mail)"
|
|
|
|
|
),
|
|
|
|
|
threshold=1,
|
|
|
|
|
evaluation_periods=3,
|
|
|
|
|
datapoints_to_alarm=2,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
|
|
|
|
|
2026-04-07 12:12:30 -04:00
|
|
|
class PoIngestStack(Stack):
|
|
|
|
|
def __init__(self, scope: Construct, construct_id: str, **kwargs):
|
|
|
|
|
super().__init__(scope, construct_id, **kwargs)
|
|
|
|
|
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
# --- Shared alarm SNS topic (site-alerts) ---
|
|
|
|
|
# Imported once near the top so every alarm in this stack reuses the same
|
|
|
|
|
# Topic construct instance (avoids duplicate logical IDs). ALARM-only
|
|
|
|
|
# SnsAction; no OK action, per the CloudWatch-alarm preference. The
|
|
|
|
|
# topic's CMK (alias/seahaven-alarm-topics) lives on the topic itself.
|
|
|
|
|
alarm_topic = sns.Topic.from_topic_arn(
|
|
|
|
|
self,
|
|
|
|
|
"SiteAlertsTopic",
|
|
|
|
|
f"arn:aws:sns:{self.region}:{self.account}:site-alerts",
|
|
|
|
|
)
|
|
|
|
|
|
2026-04-07 12:12:30 -04:00
|
|
|
# --- S3 bucket for raw emails ---
|
|
|
|
|
email_bucket = s3.Bucket(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"EmailBucket",
|
2026-04-07 12:12:30 -04:00
|
|
|
bucket_name=f"po-ingest-emails-{self.account}",
|
2026-06-17 17:23:06 -04:00
|
|
|
block_public_access=s3.BlockPublicAccess.BLOCK_ALL,
|
2026-04-07 12:12:30 -04:00
|
|
|
removal_policy=RemovalPolicy.RETAIN,
|
|
|
|
|
lifecycle_rules=[
|
|
|
|
|
s3.LifecycleRule(expiration=Duration.days(90)),
|
|
|
|
|
],
|
|
|
|
|
)
|
|
|
|
|
|
2026-06-10 19:31:55 -04:00
|
|
|
# --- Shared customer-managed CMK for sensitive DynamoDB tables ---
|
|
|
|
|
# Owned by the account-baseline app (alias/seahaven-dynamodb, INFRA-95 /
|
|
|
|
|
# M-3); ARN published to SSM. The purchase-orders table was migrated to
|
|
|
|
|
# SSE-KMS out-of-band, so declaring encryption_key here reconciles the
|
|
|
|
|
# drift and — via grant_read_write_data below — propagates the required
|
|
|
|
|
# kms:Decrypt/GenerateDataKey/DescribeKey to the consumer roles.
|
|
|
|
|
dynamodb_cmk = kms.Key.from_key_arn(
|
|
|
|
|
self,
|
|
|
|
|
"DynamoDbCmk",
|
|
|
|
|
ssm.StringParameter.value_for_string_parameter(
|
|
|
|
|
self, "/seahaven/dynamodb/cmk-arn"
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
2026-04-30 14:26:53 -04:00
|
|
|
# --- Purchase-orders DynamoDB table ---
|
|
|
|
|
# Owned by this stack. Streams enabled for the site-extractor pipeline.
|
|
|
|
|
# Other stacks (seahaven-slack-bot) reference this table via fromTableName().
|
|
|
|
|
po_table = dynamodb.Table(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"PurchaseOrdersTable",
|
2026-04-30 14:26:53 -04:00
|
|
|
table_name="purchase-orders",
|
|
|
|
|
partition_key=dynamodb.Attribute(
|
|
|
|
|
name="po_number",
|
|
|
|
|
type=dynamodb.AttributeType.STRING,
|
|
|
|
|
),
|
|
|
|
|
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
|
|
|
|
|
removal_policy=RemovalPolicy.RETAIN,
|
Align PO schema with enriched records and improve extraction prompt
Replaces extraction prompt with domain-specific rules: trade
classification taxonomy (23 categories), site_code skip list,
zip padding, revision email type, and structured extraction for
fiscal_year, trade, and coupa_category.
Handler changes:
- New "revision" email type overwrites existing PO via put_item
- enrich_parsed() adds top-level state, ship_to_raw, data_source
- pad_zip() zero-pads short zip codes (e.g., "7001" → "07001")
- Removed invoice_total/invoice_count (Payee Central only)
Web UI: added revision badge, new detail fields (site code, state,
trade, fiscal year, coupa category, data source), line item table
now shows Qty/Unit/Price columns, list view shows Site and Trade.
CDK: fixed StreamViewType to match deployed table (NEW_IMAGE).
README: documented PO record schema and revision flow.
2026-05-01 19:40:18 -04:00
|
|
|
stream=dynamodb.StreamViewType.NEW_IMAGE,
|
2026-06-10 19:31:55 -04:00
|
|
|
encryption=dynamodb.TableEncryption.CUSTOMER_MANAGED,
|
|
|
|
|
encryption_key=dynamodb_cmk,
|
2026-04-07 12:12:30 -04:00
|
|
|
)
|
|
|
|
|
|
|
|
|
|
# --- Secrets Manager for Anthropic API key ---
|
|
|
|
|
anthropic_secret = secretsmanager.Secret(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"AnthropicApiKey",
|
2026-04-07 12:12:30 -04:00
|
|
|
secret_name="po-ingest/anthropic-api-key",
|
|
|
|
|
description="Anthropic API key for Coupa PO email parsing",
|
2026-05-01 19:17:19 -04:00
|
|
|
removal_policy=RemovalPolicy.RETAIN,
|
2026-04-07 12:12:30 -04:00
|
|
|
)
|
|
|
|
|
|
Reconcile IaC with out-of-band DLQ + Function URL changes (INFRA-74, INFRA-41) (#50)
Make CDK the source of truth for two sets of changes applied out-of-band
via CLI to the po-ingest and WorkorderIngestStack stacks.
INFRA-74 (audit C-5): remove the public FunctionUrlAuthType.NONE Function
URL construct (and its auto-generated Principal:* invoke permission +
output) from both po-web-ui and workorder-web-ui. The URLs were already
deleted live via CLI; CFN's delete is idempotent.
INFRA-41 (audit H-8): add a CDK-managed SQS dead-letter queue
(dead_letter_queue=, 14d retention, SSL-enforced, CDK-generated name) and
an ALARM-only Errors alarm (Sum, threshold>0, site-alerts topic) for both
po-email-processor and workorder-email-processor, mirroring the
apm-wo-analysis-classifier DLQ and payments-payroll-batch alarm patterns.
Interim CLI resources (per-fn -dlq queues, -errors alarms, dlq-send inline
policies, OnFailure event-invoke-configs) removed post-deploy.
2026-06-08 16:02:29 -04:00
|
|
|
# --- DLQ for failed async invocations (INFRA-41 / audit H-8) ---
|
|
|
|
|
# SES → S3 → Lambda is async; without an OnFailure destination a failed
|
|
|
|
|
# parse (bad email, transient error) is silently dropped after Lambda's
|
|
|
|
|
# retries. CDK generates the queue name to avoid colliding with the
|
|
|
|
|
# interim CLI-created po-email-processor-dlq (removed post-deploy).
|
|
|
|
|
email_processor_dlq = sqs.Queue(
|
|
|
|
|
self,
|
|
|
|
|
"EmailProcessorDlq",
|
|
|
|
|
retention_period=Duration.days(14),
|
|
|
|
|
enforce_ssl=True,
|
|
|
|
|
)
|
|
|
|
|
|
2026-04-07 12:12:30 -04:00
|
|
|
# --- Lambda function ---
|
|
|
|
|
email_processor = lambda_.Function(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"EmailProcessor",
|
2026-04-07 12:12:30 -04:00
|
|
|
function_name="po-email-processor",
|
|
|
|
|
runtime=lambda_.Runtime.PYTHON_3_12,
|
2026-04-30 14:26:53 -04:00
|
|
|
architecture=lambda_.Architecture.ARM_64,
|
2026-04-07 12:12:30 -04:00
|
|
|
handler="handler.handler",
|
2026-05-08 16:01:21 -04:00
|
|
|
code=lambda_.Code.from_asset(
|
2026-05-12 15:21:06 -04:00
|
|
|
"../lambdas/po/email_processor",
|
2026-05-08 16:01:21 -04:00
|
|
|
bundling=cdk.BundlingOptions(
|
|
|
|
|
image=lambda_.Runtime.PYTHON_3_12.bundling_image,
|
|
|
|
|
command=[
|
|
|
|
|
"bash",
|
|
|
|
|
"-c",
|
2026-06-03 14:56:02 -04:00
|
|
|
"pip install --platform manylinux2014_aarch64 --only-binary=:all: "
|
|
|
|
|
"-r requirements.txt -t /asset-output && "
|
Add fail-closed SES sender authentication (INFRA-107) (#98)
* Add fail-closed SES sender authentication
The From header and any raw-MIME Authentication-Results copies are
attacker-forgeable, so a forged email to apm@int.seahaven.com or
amazon_po@int.seahaven.com could create or mutate a WO/PO (INFRA-107,
CRITICAL). Both S3-triggered email processors now authenticate the
sender against the Authentication-Results header SES itself prepends
at delivery: only the topmost header is consulted, its authserv-id
must be amazonses.com, and it must carry dkim=pass for a domain in
the per-pipeline ALLOWED_DKIM_DOMAINS env var (comma-separated, set
in CDK so ops can adjust without code changes).
Allowlists come from live traffic observed 2026-07-15 on both ingest
buckets: WO mail arrives via the apm@ Google Groups forward, which
re-signs as seahaven.com (the hxgnsmartcloud.com signature does not
survive the forward); PO mail passes for amazon.coupahost.com.
amazonses.com also passes on PO mail but is deliberately excluded --
every SES customer's outbound mail passes for it.
Every failure path (env var unset, header missing or unparseable,
verdict fail, unaligned domain) rejects the email: a structured
warning with the reason and S3 key is logged and the record skipped
without erroring the invocation, so rejected mail causes no Lambda
retries or DLQ messages. Handler signatures and event sources are
unchanged.
Refs: INFRA-107
* Harden AR parser per cross-family review
Cross-family (GPT-4.1) review findings: terminate the dkim result
token at end-of-clause, whitespace, or a comment so a value like
"dkim=pass-fake" can never be read as a pass; normalize trailing
dots off allowlist entries so "seahaven.com." matches; make the
compat32 parser policy explicit. Adds tests for result-token
boundaries, comments after the result, quoted domain values, and
folding inside a dkim clause.
Refs: INFRA-107
* Harden AR parsing and alarm on sender-auth rejects
The SES-stamped Authentication-Results value echoes attacker-controlled
SMTP-session tokens (envelope-from, helo, header.from) as their own
semicolon-delimited property clauses. A naive split(";") tore an RFC 5321
quoted-local-part MAIL FROM apart and manufactured a forged dkim=pass
clause, so a fully spoofed email was accepted on the genuinely
SES-stamped topmost header. Tokenise comment- and quoted-string-aware
(RFC 8601 / RFC 5322): strip CFWS comments, split clauses only on
semicolons outside a quoted-string, and fail closed on unbalanced
quotes/comments so a ';' inside a quoted pvalue can never start a clause.
Rejected mail returns normally (no error, no retry, no DLQ message), so a
signing-domain drift or a wrong allowlist would silently discard 100% of
legitimate mail while every alarm stayed green. Add a CloudWatch Logs
metric filter + alarm on the sender_auth_rejected warning to both stacks
so a false-reject storm pages instead of vanishing. This is also the
safety net for the WO seahaven.com allowlist assumption, which must be
validated against a live SES-stamped header (a plain Gmail auto-forward
re-signs under the sending Workspace domain, not seahaven.com).
Refs: INFRA-107
* chore: retrigger CI (no run recorded for 7c74ac1)
* Fix quoted-AUID DKIM domain spoof in sender auth
Resolve three confirmed /sh-security-review findings on the fail-closed
SES sender-authentication control.
HIGH: header.i/header.d domain extraction was not quoted-string aware.
An attacker with a valid DKIM key for their own domain could set an
RFC 6376-legal AUID such as i="@seahaven.com"@attacker.com; the naive
extractor stopped at the closing quote and returned seahaven.com,
accepting forged mail. Extraction now tokenises the clause with the same
quoted-string discipline already used for clause splitting: header.d
(the plain signing domain) is authoritative when present, otherwise the
header.i domain is the part after the AUID's LAST top-level "@", so a "@"
inside a quoted local-part is treated as signer-controlled label text and
yields the true signer (attacker.com), not seahaven.com.
LOW: the topmost-header parse ran outside evaluate_sender_authentication's
try/except, so an unexpected parser exception on crafted input could
propagate into the handler and Lambda async retries/DLQ. The parse now
fails CLOSED with an authentication_results_unparseable reason.
MEDIUM: the sender_auth_rejected alarm used Sum>=3 over 15 min, blind to
a low-volume total-reject outage (a trickle that never sums to 3). Both
stacks now alarm on >=1 reject per 5-min period with evaluation_periods=3
/ datapoints_to_alarm=2, so a sustained reject condition pages even at one
reject per period while a lone stray probe self-clears.
Refs: INFRA-107
* Load Lambda function dir on sys.path in tests
Rebasing INFRA-107 onto main folded #95's pytest suite into this
branch's tests. The unified conftest loads the PO/WO handlers by file
path, and handler.py now does `from ses_auth import
authenticate_inbound_email` -- a bare sibling import that resolves in
the Lambda only because the runtime puts each function's own directory
on sys.path. The shared load_handler now adds that directory so the
handler tests import correctly alongside the sender-auth tests.
Refs: INFRA-107
* Note #97 test files in README directory tree
The rebase onto main brought in #97's tests/requirements.txt and
tests/test_po_merge.py. List both in the directory tree so it matches
the tree on disk.
Refs: INFRA-107
* Document INFRA-107 forwarder-binding risk acceptance
Record the accepted risk that WO sender auth binds to the apm@ forward's
re-signing domain (seahaven.com) rather than the Hexagon originator; the
apm@ Google Group's restricted posting policy is the load-bearing control
(escalates to HIGH if the group is opened to external posting). Also
correct the sender-auth-rejected alarm docs to match the shipped config
(>=1 per 5-min, 2-of-3 datapoints, not the superseded >=3/15min) and
note the SES-AR-01/02 parser hardening follow-ups.
Refs: INFRA-107
2026-07-15 20:58:47 -04:00
|
|
|
"cp handler.py ses_auth.py /asset-output/",
|
2026-05-08 16:01:21 -04:00
|
|
|
],
|
|
|
|
|
),
|
|
|
|
|
),
|
2026-04-07 12:12:30 -04:00
|
|
|
timeout=Duration.seconds(60),
|
|
|
|
|
memory_size=256,
|
2026-04-30 14:26:53 -04:00
|
|
|
log_retention=logs.RetentionDays.TWO_MONTHS,
|
Reconcile IaC with out-of-band DLQ + Function URL changes (INFRA-74, INFRA-41) (#50)
Make CDK the source of truth for two sets of changes applied out-of-band
via CLI to the po-ingest and WorkorderIngestStack stacks.
INFRA-74 (audit C-5): remove the public FunctionUrlAuthType.NONE Function
URL construct (and its auto-generated Principal:* invoke permission +
output) from both po-web-ui and workorder-web-ui. The URLs were already
deleted live via CLI; CFN's delete is idempotent.
INFRA-41 (audit H-8): add a CDK-managed SQS dead-letter queue
(dead_letter_queue=, 14d retention, SSL-enforced, CDK-generated name) and
an ALARM-only Errors alarm (Sum, threshold>0, site-alerts topic) for both
po-email-processor and workorder-email-processor, mirroring the
apm-wo-analysis-classifier DLQ and payments-payroll-batch alarm patterns.
Interim CLI resources (per-fn -dlq queues, -errors alarms, dlq-send inline
policies, OnFailure event-invoke-configs) removed post-deploy.
2026-06-08 16:02:29 -04:00
|
|
|
dead_letter_queue=email_processor_dlq,
|
2026-04-07 12:12:30 -04:00
|
|
|
environment={
|
|
|
|
|
"PO_TABLE": "purchase-orders",
|
|
|
|
|
"ANTHROPIC_API_KEY_SECRET_ARN": anthropic_secret.secret_arn,
|
Add fail-closed SES sender authentication (INFRA-107) (#98)
* Add fail-closed SES sender authentication
The From header and any raw-MIME Authentication-Results copies are
attacker-forgeable, so a forged email to apm@int.seahaven.com or
amazon_po@int.seahaven.com could create or mutate a WO/PO (INFRA-107,
CRITICAL). Both S3-triggered email processors now authenticate the
sender against the Authentication-Results header SES itself prepends
at delivery: only the topmost header is consulted, its authserv-id
must be amazonses.com, and it must carry dkim=pass for a domain in
the per-pipeline ALLOWED_DKIM_DOMAINS env var (comma-separated, set
in CDK so ops can adjust without code changes).
Allowlists come from live traffic observed 2026-07-15 on both ingest
buckets: WO mail arrives via the apm@ Google Groups forward, which
re-signs as seahaven.com (the hxgnsmartcloud.com signature does not
survive the forward); PO mail passes for amazon.coupahost.com.
amazonses.com also passes on PO mail but is deliberately excluded --
every SES customer's outbound mail passes for it.
Every failure path (env var unset, header missing or unparseable,
verdict fail, unaligned domain) rejects the email: a structured
warning with the reason and S3 key is logged and the record skipped
without erroring the invocation, so rejected mail causes no Lambda
retries or DLQ messages. Handler signatures and event sources are
unchanged.
Refs: INFRA-107
* Harden AR parser per cross-family review
Cross-family (GPT-4.1) review findings: terminate the dkim result
token at end-of-clause, whitespace, or a comment so a value like
"dkim=pass-fake" can never be read as a pass; normalize trailing
dots off allowlist entries so "seahaven.com." matches; make the
compat32 parser policy explicit. Adds tests for result-token
boundaries, comments after the result, quoted domain values, and
folding inside a dkim clause.
Refs: INFRA-107
* Harden AR parsing and alarm on sender-auth rejects
The SES-stamped Authentication-Results value echoes attacker-controlled
SMTP-session tokens (envelope-from, helo, header.from) as their own
semicolon-delimited property clauses. A naive split(";") tore an RFC 5321
quoted-local-part MAIL FROM apart and manufactured a forged dkim=pass
clause, so a fully spoofed email was accepted on the genuinely
SES-stamped topmost header. Tokenise comment- and quoted-string-aware
(RFC 8601 / RFC 5322): strip CFWS comments, split clauses only on
semicolons outside a quoted-string, and fail closed on unbalanced
quotes/comments so a ';' inside a quoted pvalue can never start a clause.
Rejected mail returns normally (no error, no retry, no DLQ message), so a
signing-domain drift or a wrong allowlist would silently discard 100% of
legitimate mail while every alarm stayed green. Add a CloudWatch Logs
metric filter + alarm on the sender_auth_rejected warning to both stacks
so a false-reject storm pages instead of vanishing. This is also the
safety net for the WO seahaven.com allowlist assumption, which must be
validated against a live SES-stamped header (a plain Gmail auto-forward
re-signs under the sending Workspace domain, not seahaven.com).
Refs: INFRA-107
* chore: retrigger CI (no run recorded for 7c74ac1)
* Fix quoted-AUID DKIM domain spoof in sender auth
Resolve three confirmed /sh-security-review findings on the fail-closed
SES sender-authentication control.
HIGH: header.i/header.d domain extraction was not quoted-string aware.
An attacker with a valid DKIM key for their own domain could set an
RFC 6376-legal AUID such as i="@seahaven.com"@attacker.com; the naive
extractor stopped at the closing quote and returned seahaven.com,
accepting forged mail. Extraction now tokenises the clause with the same
quoted-string discipline already used for clause splitting: header.d
(the plain signing domain) is authoritative when present, otherwise the
header.i domain is the part after the AUID's LAST top-level "@", so a "@"
inside a quoted local-part is treated as signer-controlled label text and
yields the true signer (attacker.com), not seahaven.com.
LOW: the topmost-header parse ran outside evaluate_sender_authentication's
try/except, so an unexpected parser exception on crafted input could
propagate into the handler and Lambda async retries/DLQ. The parse now
fails CLOSED with an authentication_results_unparseable reason.
MEDIUM: the sender_auth_rejected alarm used Sum>=3 over 15 min, blind to
a low-volume total-reject outage (a trickle that never sums to 3). Both
stacks now alarm on >=1 reject per 5-min period with evaluation_periods=3
/ datapoints_to_alarm=2, so a sustained reject condition pages even at one
reject per period while a lone stray probe self-clears.
Refs: INFRA-107
* Load Lambda function dir on sys.path in tests
Rebasing INFRA-107 onto main folded #95's pytest suite into this
branch's tests. The unified conftest loads the PO/WO handlers by file
path, and handler.py now does `from ses_auth import
authenticate_inbound_email` -- a bare sibling import that resolves in
the Lambda only because the runtime puts each function's own directory
on sys.path. The shared load_handler now adds that directory so the
handler tests import correctly alongside the sender-auth tests.
Refs: INFRA-107
* Note #97 test files in README directory tree
The rebase onto main brought in #97's tests/requirements.txt and
tests/test_po_merge.py. List both in the directory tree so it matches
the tree on disk.
Refs: INFRA-107
* Document INFRA-107 forwarder-binding risk acceptance
Record the accepted risk that WO sender auth binds to the apm@ forward's
re-signing domain (seahaven.com) rather than the Hexagon originator; the
apm@ Google Group's restricted posting policy is the load-bearing control
(escalates to HIGH if the group is opened to external posting). Also
correct the sender-auth-rejected alarm docs to match the shipped config
(>=1 per 5-min, 2-of-3 datapoints, not the superseded >=3/15min) and
note the SES-AR-01/02 parser hardening follow-ups.
Refs: INFRA-107
2026-07-15 20:58:47 -04:00
|
|
|
# Fail-closed sender auth (INFRA-107): the handler only
|
|
|
|
|
# accepts mail whose SES-stamped Authentication-Results
|
|
|
|
|
# header carries dkim=pass for one of these domains.
|
|
|
|
|
# Observed on live traffic 2026-07-15: Coupa PO mail passes
|
|
|
|
|
# DKIM for amazon.coupahost.com (and amazonses.com, which is
|
|
|
|
|
# deliberately NOT allowlisted — every SES customer's mail
|
|
|
|
|
# passes that). Unset/empty ⇒ the handler rejects all mail.
|
|
|
|
|
"ALLOWED_DKIM_DOMAINS": "amazon.coupahost.com",
|
2026-04-07 12:12:30 -04:00
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
# Grant permissions
|
|
|
|
|
email_bucket.grant_read(email_processor)
|
|
|
|
|
po_table.grant_read_write_data(email_processor)
|
|
|
|
|
anthropic_secret.grant_read(email_processor)
|
|
|
|
|
|
Reconcile IaC with out-of-band DLQ + Function URL changes (INFRA-74, INFRA-41) (#50)
Make CDK the source of truth for two sets of changes applied out-of-band
via CLI to the po-ingest and WorkorderIngestStack stacks.
INFRA-74 (audit C-5): remove the public FunctionUrlAuthType.NONE Function
URL construct (and its auto-generated Principal:* invoke permission +
output) from both po-web-ui and workorder-web-ui. The URLs were already
deleted live via CLI; CFN's delete is idempotent.
INFRA-41 (audit H-8): add a CDK-managed SQS dead-letter queue
(dead_letter_queue=, 14d retention, SSL-enforced, CDK-generated name) and
an ALARM-only Errors alarm (Sum, threshold>0, site-alerts topic) for both
po-email-processor and workorder-email-processor, mirroring the
apm-wo-analysis-classifier DLQ and payments-payroll-batch alarm patterns.
Interim CLI resources (per-fn -dlq queues, -errors alarms, dlq-send inline
policies, OnFailure event-invoke-configs) removed post-deploy.
2026-06-08 16:02:29 -04:00
|
|
|
# --- Errors alarm (INFRA-41 / audit H-8) ---
|
|
|
|
|
# ALARM-only (no OK action, per the CloudWatch-alarm preference) to the
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
# shared site-alerts topic. Any errored invocation in a 5-min window pages.
|
Reconcile IaC with out-of-band DLQ + Function URL changes (INFRA-74, INFRA-41) (#50)
Make CDK the source of truth for two sets of changes applied out-of-band
via CLI to the po-ingest and WorkorderIngestStack stacks.
INFRA-74 (audit C-5): remove the public FunctionUrlAuthType.NONE Function
URL construct (and its auto-generated Principal:* invoke permission +
output) from both po-web-ui and workorder-web-ui. The URLs were already
deleted live via CLI; CFN's delete is idempotent.
INFRA-41 (audit H-8): add a CDK-managed SQS dead-letter queue
(dead_letter_queue=, 14d retention, SSL-enforced, CDK-generated name) and
an ALARM-only Errors alarm (Sum, threshold>0, site-alerts topic) for both
po-email-processor and workorder-email-processor, mirroring the
apm-wo-analysis-classifier DLQ and payments-payroll-batch alarm patterns.
Interim CLI resources (per-fn -dlq queues, -errors alarms, dlq-send inline
policies, OnFailure event-invoke-configs) removed post-deploy.
2026-06-08 16:02:29 -04:00
|
|
|
email_processor.metric_errors(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"EmailProcessorErrorsAlarm",
|
|
|
|
|
alarm_name="po-email-processor-errors",
|
|
|
|
|
alarm_description="po-email-processor async invocation errors",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
Add fail-closed SES sender authentication (INFRA-107) (#98)
* Add fail-closed SES sender authentication
The From header and any raw-MIME Authentication-Results copies are
attacker-forgeable, so a forged email to apm@int.seahaven.com or
amazon_po@int.seahaven.com could create or mutate a WO/PO (INFRA-107,
CRITICAL). Both S3-triggered email processors now authenticate the
sender against the Authentication-Results header SES itself prepends
at delivery: only the topmost header is consulted, its authserv-id
must be amazonses.com, and it must carry dkim=pass for a domain in
the per-pipeline ALLOWED_DKIM_DOMAINS env var (comma-separated, set
in CDK so ops can adjust without code changes).
Allowlists come from live traffic observed 2026-07-15 on both ingest
buckets: WO mail arrives via the apm@ Google Groups forward, which
re-signs as seahaven.com (the hxgnsmartcloud.com signature does not
survive the forward); PO mail passes for amazon.coupahost.com.
amazonses.com also passes on PO mail but is deliberately excluded --
every SES customer's outbound mail passes for it.
Every failure path (env var unset, header missing or unparseable,
verdict fail, unaligned domain) rejects the email: a structured
warning with the reason and S3 key is logged and the record skipped
without erroring the invocation, so rejected mail causes no Lambda
retries or DLQ messages. Handler signatures and event sources are
unchanged.
Refs: INFRA-107
* Harden AR parser per cross-family review
Cross-family (GPT-4.1) review findings: terminate the dkim result
token at end-of-clause, whitespace, or a comment so a value like
"dkim=pass-fake" can never be read as a pass; normalize trailing
dots off allowlist entries so "seahaven.com." matches; make the
compat32 parser policy explicit. Adds tests for result-token
boundaries, comments after the result, quoted domain values, and
folding inside a dkim clause.
Refs: INFRA-107
* Harden AR parsing and alarm on sender-auth rejects
The SES-stamped Authentication-Results value echoes attacker-controlled
SMTP-session tokens (envelope-from, helo, header.from) as their own
semicolon-delimited property clauses. A naive split(";") tore an RFC 5321
quoted-local-part MAIL FROM apart and manufactured a forged dkim=pass
clause, so a fully spoofed email was accepted on the genuinely
SES-stamped topmost header. Tokenise comment- and quoted-string-aware
(RFC 8601 / RFC 5322): strip CFWS comments, split clauses only on
semicolons outside a quoted-string, and fail closed on unbalanced
quotes/comments so a ';' inside a quoted pvalue can never start a clause.
Rejected mail returns normally (no error, no retry, no DLQ message), so a
signing-domain drift or a wrong allowlist would silently discard 100% of
legitimate mail while every alarm stayed green. Add a CloudWatch Logs
metric filter + alarm on the sender_auth_rejected warning to both stacks
so a false-reject storm pages instead of vanishing. This is also the
safety net for the WO seahaven.com allowlist assumption, which must be
validated against a live SES-stamped header (a plain Gmail auto-forward
re-signs under the sending Workspace domain, not seahaven.com).
Refs: INFRA-107
* chore: retrigger CI (no run recorded for 7c74ac1)
* Fix quoted-AUID DKIM domain spoof in sender auth
Resolve three confirmed /sh-security-review findings on the fail-closed
SES sender-authentication control.
HIGH: header.i/header.d domain extraction was not quoted-string aware.
An attacker with a valid DKIM key for their own domain could set an
RFC 6376-legal AUID such as i="@seahaven.com"@attacker.com; the naive
extractor stopped at the closing quote and returned seahaven.com,
accepting forged mail. Extraction now tokenises the clause with the same
quoted-string discipline already used for clause splitting: header.d
(the plain signing domain) is authoritative when present, otherwise the
header.i domain is the part after the AUID's LAST top-level "@", so a "@"
inside a quoted local-part is treated as signer-controlled label text and
yields the true signer (attacker.com), not seahaven.com.
LOW: the topmost-header parse ran outside evaluate_sender_authentication's
try/except, so an unexpected parser exception on crafted input could
propagate into the handler and Lambda async retries/DLQ. The parse now
fails CLOSED with an authentication_results_unparseable reason.
MEDIUM: the sender_auth_rejected alarm used Sum>=3 over 15 min, blind to
a low-volume total-reject outage (a trickle that never sums to 3). Both
stacks now alarm on >=1 reject per 5-min period with evaluation_periods=3
/ datapoints_to_alarm=2, so a sustained reject condition pages even at one
reject per period while a lone stray probe self-clears.
Refs: INFRA-107
* Load Lambda function dir on sys.path in tests
Rebasing INFRA-107 onto main folded #95's pytest suite into this
branch's tests. The unified conftest loads the PO/WO handlers by file
path, and handler.py now does `from ses_auth import
authenticate_inbound_email` -- a bare sibling import that resolves in
the Lambda only because the runtime puts each function's own directory
on sys.path. The shared load_handler now adds that directory so the
handler tests import correctly alongside the sender-auth tests.
Refs: INFRA-107
* Note #97 test files in README directory tree
The rebase onto main brought in #97's tests/requirements.txt and
tests/test_po_merge.py. List both in the directory tree so it matches
the tree on disk.
Refs: INFRA-107
* Document INFRA-107 forwarder-binding risk acceptance
Record the accepted risk that WO sender auth binds to the apm@ forward's
re-signing domain (seahaven.com) rather than the Hexagon originator; the
apm@ Google Group's restricted posting policy is the load-bearing control
(escalates to HIGH if the group is opened to external posting). Also
correct the sender-auth-rejected alarm docs to match the shipped config
(>=1 per 5-min, 2-of-3 datapoints, not the superseded >=3/15min) and
note the SES-AR-01/02 parser hardening follow-ups.
Refs: INFRA-107
2026-07-15 20:58:47 -04:00
|
|
|
# --- Sender-auth rejection alarm (INFRA-107) ---
|
|
|
|
|
# A rejected email (bad/unaligned DKIM verdict) returns normally, so it
|
|
|
|
|
# produces NO Lambda error, NO DLQ message and NO retry -- only a
|
|
|
|
|
# `sender_auth_rejected` warning log. Without this metric filter + alarm a
|
|
|
|
|
# domain drift (Coupa rotates its signing subdomain, SES changes its
|
|
|
|
|
# Authentication-Results format, the allowlist is wrong) would silently
|
|
|
|
|
# discard 100% of legitimate PO mail while every other alarm stays green.
|
|
|
|
|
# A CloudWatch Logs metric filter turns those warnings into a metric so a
|
|
|
|
|
# false-reject storm pages instead of vanishing. default_value=0 keeps the
|
|
|
|
|
# series populated (alarm stays OK, never INSUFFICIENT_DATA) between events.
|
|
|
|
|
_add_sender_auth_rejected_alarm(
|
|
|
|
|
self, "EmailProcessor", "po-email-processor", alarm_topic
|
|
|
|
|
)
|
|
|
|
|
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
# --- Throttles alarm: po-email-processor ---
|
|
|
|
|
# Any throttled invocation (concurrency cap hit) in a 5-min window pages.
|
|
|
|
|
# ALARM-only to site-alerts; no OK action; NOT_BREACHING when no data.
|
|
|
|
|
email_processor.metric_throttles(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"EmailProcessorThrottlesAlarm",
|
|
|
|
|
alarm_name="po-email-processor-throttles",
|
|
|
|
|
alarm_description="po-email-processor invocation throttles",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
2026-06-17 17:28:42 -04:00
|
|
|
# --- DLQ messages-present alarm ---
|
|
|
|
|
# Pages when any message lands in the EmailProcessorDlq: a message here
|
|
|
|
|
# means a PO email was permanently dropped after Lambda exhausted its
|
|
|
|
|
# async retries. Maximum over a single 5-min window > 0 fires; missing
|
|
|
|
|
# data (no messages metric emitted) is not breaching. Reuses the shared
|
|
|
|
|
# site-alerts topic, ALARM-only, like the errors alarm above. The metric
|
|
|
|
|
# helper derives the QueueName dimension from the queue construct, so the
|
|
|
|
|
# alarm tracks the CDK-generated queue name without hardcoding it.
|
|
|
|
|
email_processor_dlq.metric_approximate_number_of_messages_visible(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Maximum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"EmailProcessorDlqMessagesAlarm",
|
|
|
|
|
alarm_name="po-email-processor-dlq-messages",
|
|
|
|
|
alarm_description="po-email-processor DLQ has messages (dropped PO emails)",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
# --- Duration alarm: po-email-processor (orphan adoption) ---
|
|
|
|
|
# Adopts the orphaned CLI alarm Lambda-Duration-po-email-processor under
|
|
|
|
|
# the repo's <fn>-duration naming (NEW logical name → no deploy collision;
|
|
|
|
|
# delete the orphan post-deploy). p99 / 45000 ms
|
|
|
|
|
# (75% of the 60s timeout) / eval 3 of 3 — tighter than the orphan's
|
|
|
|
|
# Maximum>=48000 / 1-of-1.
|
|
|
|
|
email_processor.metric_duration(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="p99",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"EmailProcessorDurationAlarm",
|
|
|
|
|
alarm_name="po-email-processor-duration",
|
|
|
|
|
alarm_description="po-email-processor p99 duration approaching the 60s timeout",
|
|
|
|
|
threshold=45000,
|
|
|
|
|
evaluation_periods=3,
|
|
|
|
|
datapoints_to_alarm=2,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
2026-04-07 12:12:30 -04:00
|
|
|
# S3 event notification → Lambda
|
|
|
|
|
email_bucket.add_event_notification(
|
|
|
|
|
s3.EventType.OBJECT_CREATED,
|
|
|
|
|
s3n.LambdaDestination(email_processor),
|
|
|
|
|
s3.NotificationKeyFilter(prefix="inbound/"),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
# --- SES Receipt Rule ---
|
|
|
|
|
# Reuse the existing INBOUND_MAIL rule set (shared with workorder-ingest)
|
|
|
|
|
rule_set = ses.ReceiptRuleSet.from_receipt_rule_set_name(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"ExistingRuleSet",
|
|
|
|
|
"INBOUND_MAIL",
|
2026-04-07 12:12:30 -04:00
|
|
|
)
|
|
|
|
|
|
|
|
|
|
rule_set.add_rule(
|
|
|
|
|
"PoEmailRule",
|
|
|
|
|
recipients=["amazon_po@int.seahaven.com"],
|
|
|
|
|
actions=[
|
|
|
|
|
ses_actions.S3(
|
|
|
|
|
bucket=email_bucket,
|
|
|
|
|
object_key_prefix="inbound/",
|
|
|
|
|
),
|
|
|
|
|
],
|
|
|
|
|
)
|
|
|
|
|
|
Land safe fixes from 2026-06-17 security sweep (#97)
* Remove gratuitous KMS grant on shared DynamoDB CMK
wo-email-processor held grant_encrypt_decrypt on the shared
seahaven-dynamodb CMK, but the WorkOrders/WorkOrderComments tables
are not encrypted with that CMK. The grant was dead weight that
extended the WO processor's decrypt reach to the CMK protecting the
purchase-orders table (cross-stack decrypt). Drop it to restore
least privilege; re-add as part of the table CMK migration (INFRA-6).
Refs: INFRA-6
* Require Secrets Manager key for Anthropic client
Remove the silent fallback to a plaintext ANTHROPIC_API_KEY env var
in both email processors; require ANTHROPIC_API_KEY_SECRET_ARN and
raise if absent so a misconfigured deploy fails loudly instead of
using an unmanaged key.
Adapted from f175323 on security/sweep-2026-06-17. The From-header
sender-domain allowlist from that commit is intentionally dropped:
the From header is spoofable (INFRA-107, confirmed critical) and
sender authentication is being reworked in a separate PR.
Refs: INFRA-107
* Merge PO revisions and handle out-of-order events
save_revision did a full put_item overwrite, so a revision omitting
line_items/supplier permanently deleted them. save_new_po used a
conditional put that silently dropped the PO when an out-of-order
cancellation had already created a skeleton row.
Switch both to field-level merge update_items: a revision now SETs
only the fields it carries, and a new_po backfills data into a
pre-existing Cancelled skeleton while preserving the Cancelled
status. No email can now delete data established by an earlier one.
* Gate web UIs behind auth and escape currency XSS
The po-web-ui and workorder-web-ui handlers had no auth: any
invocation path returned the full PO/WO DB. Add a fail-closed
shared-secret gate (X-Auth-Token / Bearer, constant-time compared to
WEB_UI_AUTH_TOKEN) so a future re-attached Function URL cannot
re-expose the data (URLs removed under INFRA-74). Wire the token from
the SSM String param /procurement-ingest/web-ui-auth-token.
Also fix stored XSS in po-web-ui fmt_currency: the non-numeric
fallback returned str(val) unescaped, so a prompt-injected email
could make Claude emit total_amount as <script>. Escape it.
Refs: INFRA-74
* Document sweep security fixes and merge semantics
Update the README for the 2026-06-17 security sweep: required
Secrets Manager key (no plaintext env fallback), web UI auth gate +
SSM token setup step, output-escaping note, and the new PO
revision/cancellation merge behavior.
Adapted from d91f45e on security/sweep-2026-06-17; the sender
allowlist documentation is dropped along with the allowlist itself
(deferred to the INFRA-107 sender-authentication rework).
Refs: INFRA-107
* fix: resolve web UI auth token from Secrets Manager at runtime
Replace the plaintext SSM String parameter with a Secrets Manager secret
referenced by ARN only. The token is fetched and cached at module level on
first invocation, keeping shared secrets out of CloudFormation templates and
Lambda environment variables.
Refs: PR-97
* Add TTL to web UI auth token cache for rotation
The web-ui handlers cached the Secrets Manager auth token at module
level with no expiry, so a rotated secret was only picked up when the
warm container recycled — an emergency rotation could take hours to
take effect. Cache the fetched value for a 5-minute TTL instead, so a
rotated token propagates within the TTL while still avoiding a Secrets
Manager call on every request. Still fails closed when the secret is
unset or unreadable.
Refs: INFRA-74
* Log Secrets Manager failures in web UI auth token fetch
The web UI auth gate correctly fails closed when the shared token
cannot be read, but _get_auth_token() swallowed every exception
silently. A Secrets Manager permission or config error then made
every request 401 with no operational signal, leaving an outage
indistinguishable from ordinary unauthenticated traffic.
Add a module-level logger to both web_ui handlers and log the
fetch failure with logger.exception() in the except block before
returning None. Behavior is unchanged (still fails closed); the
failure is now visible in CloudWatch. The secret value is never
logged. The two handlers stay byte-consistent in the mirrored
_get_auth_token() region.
The companion finding on the CDK import of the shared
procurement-ingest/web-ui-auth-token secret was evaluated and left
as-is: the token is a single secret shared by both the PO and WO
stacks, so from_secret_name_v2 (which scopes grant_read via the
standard 6-char suffix wildcard) is correct; making it a CDK-managed
Secret in both stacks would collide the two stacks on the same
explicit secret name at deploy time.
Refs: INFRA-74
* Make Cancelled PO status sticky via atomic write
The PO merge path read status with a get_item (_is_cancelled) and then
wrote with an unconditional update_item. Two defects followed from this:
- Race (Issue A): a cancellation landing between the read and the write
was silently un-cancelled by a revision carrying a non-cancelled
po_status — a TOCTOU on a table with concurrent email processing.
- Over-broad strip (Issue B): save_revision dropped po_status whenever
the PO was Cancelled, so legitimate status updates on non-cancelled
POs and status-less revisions were affected rather than only the true
un-cancel transition.
Enforce the invariant server-side instead. "Cancelled" is a sticky,
authoritative status: once set, later new_po/revision emails may enrich
other fields but must never move it to a non-cancelled status. When the
payload carries a non-cancelled po_status, _merge_update issues the
update_item guarded by ConditionExpression "attribute_not_exists(po_status)
OR po_status <> :marker", evaluated atomically at write time, so a
cancellation that lands first always wins. On ConditionalCheckFailedException
the same fields are re-written without po_status/cancelled_at, enriching the
record while Cancelled sticks. Payloads with no status change, or an already
-Cancelled status, take a plain merge — the status is only ever suppressed on
a real un-cancel. This removes the non-atomic get_item from the write path;
_is_cancelled is deleted. Key schema and attribute names are unchanged, so the
cross-stack purchase-orders contract (read-only by seahaven-slack-bot) holds.
Add moto-backed tests covering un-cancel suppression with field enrichment,
status-less merge onto a Cancelled PO, legitimate status updates on
non-cancelled POs, new_po backfill of a Cancelled skeleton, fresh
create/merge, and authoritative save_cancellation.
Refs: #97
2026-07-15 20:17:46 -04:00
|
|
|
# --- Web UI auth token secret ---
|
|
|
|
|
# Shared secret for the web UI auth gate, stored in Secrets Manager and
|
|
|
|
|
# resolved at runtime so the token never appears in CloudFormation templates
|
|
|
|
|
# or Lambda environment variables. Create this secret before deploying
|
|
|
|
|
# either stack; both PO and WO stacks reference it by name.
|
|
|
|
|
web_ui_auth_secret = secretsmanager.Secret.from_secret_name_v2(
|
|
|
|
|
self,
|
|
|
|
|
"WebUiAuthToken",
|
|
|
|
|
"procurement-ingest/web-ui-auth-token",
|
|
|
|
|
)
|
|
|
|
|
|
2026-04-07 12:12:30 -04:00
|
|
|
# --- Web UI Lambda ---
|
|
|
|
|
web_ui = lambda_.Function(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"WebUI",
|
2026-04-07 12:12:30 -04:00
|
|
|
function_name="po-web-ui",
|
|
|
|
|
runtime=lambda_.Runtime.PYTHON_3_12,
|
2026-04-30 14:26:53 -04:00
|
|
|
architecture=lambda_.Architecture.ARM_64,
|
2026-04-07 12:12:30 -04:00
|
|
|
handler="handler.handler",
|
2026-05-12 15:21:06 -04:00
|
|
|
code=lambda_.Code.from_asset("../lambdas/po/web_ui"),
|
2026-04-07 12:12:30 -04:00
|
|
|
timeout=Duration.seconds(60),
|
|
|
|
|
memory_size=256,
|
2026-04-30 14:26:53 -04:00
|
|
|
log_retention=logs.RetentionDays.TWO_MONTHS,
|
2026-04-07 12:12:30 -04:00
|
|
|
environment={
|
|
|
|
|
"PO_TABLE": "purchase-orders",
|
Land safe fixes from 2026-06-17 security sweep (#97)
* Remove gratuitous KMS grant on shared DynamoDB CMK
wo-email-processor held grant_encrypt_decrypt on the shared
seahaven-dynamodb CMK, but the WorkOrders/WorkOrderComments tables
are not encrypted with that CMK. The grant was dead weight that
extended the WO processor's decrypt reach to the CMK protecting the
purchase-orders table (cross-stack decrypt). Drop it to restore
least privilege; re-add as part of the table CMK migration (INFRA-6).
Refs: INFRA-6
* Require Secrets Manager key for Anthropic client
Remove the silent fallback to a plaintext ANTHROPIC_API_KEY env var
in both email processors; require ANTHROPIC_API_KEY_SECRET_ARN and
raise if absent so a misconfigured deploy fails loudly instead of
using an unmanaged key.
Adapted from f175323 on security/sweep-2026-06-17. The From-header
sender-domain allowlist from that commit is intentionally dropped:
the From header is spoofable (INFRA-107, confirmed critical) and
sender authentication is being reworked in a separate PR.
Refs: INFRA-107
* Merge PO revisions and handle out-of-order events
save_revision did a full put_item overwrite, so a revision omitting
line_items/supplier permanently deleted them. save_new_po used a
conditional put that silently dropped the PO when an out-of-order
cancellation had already created a skeleton row.
Switch both to field-level merge update_items: a revision now SETs
only the fields it carries, and a new_po backfills data into a
pre-existing Cancelled skeleton while preserving the Cancelled
status. No email can now delete data established by an earlier one.
* Gate web UIs behind auth and escape currency XSS
The po-web-ui and workorder-web-ui handlers had no auth: any
invocation path returned the full PO/WO DB. Add a fail-closed
shared-secret gate (X-Auth-Token / Bearer, constant-time compared to
WEB_UI_AUTH_TOKEN) so a future re-attached Function URL cannot
re-expose the data (URLs removed under INFRA-74). Wire the token from
the SSM String param /procurement-ingest/web-ui-auth-token.
Also fix stored XSS in po-web-ui fmt_currency: the non-numeric
fallback returned str(val) unescaped, so a prompt-injected email
could make Claude emit total_amount as <script>. Escape it.
Refs: INFRA-74
* Document sweep security fixes and merge semantics
Update the README for the 2026-06-17 security sweep: required
Secrets Manager key (no plaintext env fallback), web UI auth gate +
SSM token setup step, output-escaping note, and the new PO
revision/cancellation merge behavior.
Adapted from d91f45e on security/sweep-2026-06-17; the sender
allowlist documentation is dropped along with the allowlist itself
(deferred to the INFRA-107 sender-authentication rework).
Refs: INFRA-107
* fix: resolve web UI auth token from Secrets Manager at runtime
Replace the plaintext SSM String parameter with a Secrets Manager secret
referenced by ARN only. The token is fetched and cached at module level on
first invocation, keeping shared secrets out of CloudFormation templates and
Lambda environment variables.
Refs: PR-97
* Add TTL to web UI auth token cache for rotation
The web-ui handlers cached the Secrets Manager auth token at module
level with no expiry, so a rotated secret was only picked up when the
warm container recycled — an emergency rotation could take hours to
take effect. Cache the fetched value for a 5-minute TTL instead, so a
rotated token propagates within the TTL while still avoiding a Secrets
Manager call on every request. Still fails closed when the secret is
unset or unreadable.
Refs: INFRA-74
* Log Secrets Manager failures in web UI auth token fetch
The web UI auth gate correctly fails closed when the shared token
cannot be read, but _get_auth_token() swallowed every exception
silently. A Secrets Manager permission or config error then made
every request 401 with no operational signal, leaving an outage
indistinguishable from ordinary unauthenticated traffic.
Add a module-level logger to both web_ui handlers and log the
fetch failure with logger.exception() in the except block before
returning None. Behavior is unchanged (still fails closed); the
failure is now visible in CloudWatch. The secret value is never
logged. The two handlers stay byte-consistent in the mirrored
_get_auth_token() region.
The companion finding on the CDK import of the shared
procurement-ingest/web-ui-auth-token secret was evaluated and left
as-is: the token is a single secret shared by both the PO and WO
stacks, so from_secret_name_v2 (which scopes grant_read via the
standard 6-char suffix wildcard) is correct; making it a CDK-managed
Secret in both stacks would collide the two stacks on the same
explicit secret name at deploy time.
Refs: INFRA-74
* Make Cancelled PO status sticky via atomic write
The PO merge path read status with a get_item (_is_cancelled) and then
wrote with an unconditional update_item. Two defects followed from this:
- Race (Issue A): a cancellation landing between the read and the write
was silently un-cancelled by a revision carrying a non-cancelled
po_status — a TOCTOU on a table with concurrent email processing.
- Over-broad strip (Issue B): save_revision dropped po_status whenever
the PO was Cancelled, so legitimate status updates on non-cancelled
POs and status-less revisions were affected rather than only the true
un-cancel transition.
Enforce the invariant server-side instead. "Cancelled" is a sticky,
authoritative status: once set, later new_po/revision emails may enrich
other fields but must never move it to a non-cancelled status. When the
payload carries a non-cancelled po_status, _merge_update issues the
update_item guarded by ConditionExpression "attribute_not_exists(po_status)
OR po_status <> :marker", evaluated atomically at write time, so a
cancellation that lands first always wins. On ConditionalCheckFailedException
the same fields are re-written without po_status/cancelled_at, enriching the
record while Cancelled sticks. Payloads with no status change, or an already
-Cancelled status, take a plain merge — the status is only ever suppressed on
a real un-cancel. This removes the non-atomic get_item from the write path;
_is_cancelled is deleted. Key schema and attribute names are unchanged, so the
cross-stack purchase-orders contract (read-only by seahaven-slack-bot) holds.
Add moto-backed tests covering un-cancel suppression with field enrichment,
status-less merge onto a Cancelled PO, legitimate status updates on
non-cancelled POs, new_po backfill of a Cancelled skeleton, fresh
create/merge, and authoritative save_cancellation.
Refs: #97
2026-07-15 20:17:46 -04:00
|
|
|
# Defense-in-depth shared secret for the web UI handler. The
|
|
|
|
|
# handler fails closed if this ARN is unset or the secret is
|
|
|
|
|
# missing, so any future invocation path cannot re-expose the
|
|
|
|
|
# PO DB unauthenticated. The secret value is fetched at runtime
|
|
|
|
|
# from Secrets Manager (not embedded in env vars or template).
|
|
|
|
|
"WEB_UI_AUTH_TOKEN_SECRET_ARN": web_ui_auth_secret.secret_arn,
|
2026-04-07 12:12:30 -04:00
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
po_table.grant_read_data(web_ui)
|
Land safe fixes from 2026-06-17 security sweep (#97)
* Remove gratuitous KMS grant on shared DynamoDB CMK
wo-email-processor held grant_encrypt_decrypt on the shared
seahaven-dynamodb CMK, but the WorkOrders/WorkOrderComments tables
are not encrypted with that CMK. The grant was dead weight that
extended the WO processor's decrypt reach to the CMK protecting the
purchase-orders table (cross-stack decrypt). Drop it to restore
least privilege; re-add as part of the table CMK migration (INFRA-6).
Refs: INFRA-6
* Require Secrets Manager key for Anthropic client
Remove the silent fallback to a plaintext ANTHROPIC_API_KEY env var
in both email processors; require ANTHROPIC_API_KEY_SECRET_ARN and
raise if absent so a misconfigured deploy fails loudly instead of
using an unmanaged key.
Adapted from f175323 on security/sweep-2026-06-17. The From-header
sender-domain allowlist from that commit is intentionally dropped:
the From header is spoofable (INFRA-107, confirmed critical) and
sender authentication is being reworked in a separate PR.
Refs: INFRA-107
* Merge PO revisions and handle out-of-order events
save_revision did a full put_item overwrite, so a revision omitting
line_items/supplier permanently deleted them. save_new_po used a
conditional put that silently dropped the PO when an out-of-order
cancellation had already created a skeleton row.
Switch both to field-level merge update_items: a revision now SETs
only the fields it carries, and a new_po backfills data into a
pre-existing Cancelled skeleton while preserving the Cancelled
status. No email can now delete data established by an earlier one.
* Gate web UIs behind auth and escape currency XSS
The po-web-ui and workorder-web-ui handlers had no auth: any
invocation path returned the full PO/WO DB. Add a fail-closed
shared-secret gate (X-Auth-Token / Bearer, constant-time compared to
WEB_UI_AUTH_TOKEN) so a future re-attached Function URL cannot
re-expose the data (URLs removed under INFRA-74). Wire the token from
the SSM String param /procurement-ingest/web-ui-auth-token.
Also fix stored XSS in po-web-ui fmt_currency: the non-numeric
fallback returned str(val) unescaped, so a prompt-injected email
could make Claude emit total_amount as <script>. Escape it.
Refs: INFRA-74
* Document sweep security fixes and merge semantics
Update the README for the 2026-06-17 security sweep: required
Secrets Manager key (no plaintext env fallback), web UI auth gate +
SSM token setup step, output-escaping note, and the new PO
revision/cancellation merge behavior.
Adapted from d91f45e on security/sweep-2026-06-17; the sender
allowlist documentation is dropped along with the allowlist itself
(deferred to the INFRA-107 sender-authentication rework).
Refs: INFRA-107
* fix: resolve web UI auth token from Secrets Manager at runtime
Replace the plaintext SSM String parameter with a Secrets Manager secret
referenced by ARN only. The token is fetched and cached at module level on
first invocation, keeping shared secrets out of CloudFormation templates and
Lambda environment variables.
Refs: PR-97
* Add TTL to web UI auth token cache for rotation
The web-ui handlers cached the Secrets Manager auth token at module
level with no expiry, so a rotated secret was only picked up when the
warm container recycled — an emergency rotation could take hours to
take effect. Cache the fetched value for a 5-minute TTL instead, so a
rotated token propagates within the TTL while still avoiding a Secrets
Manager call on every request. Still fails closed when the secret is
unset or unreadable.
Refs: INFRA-74
* Log Secrets Manager failures in web UI auth token fetch
The web UI auth gate correctly fails closed when the shared token
cannot be read, but _get_auth_token() swallowed every exception
silently. A Secrets Manager permission or config error then made
every request 401 with no operational signal, leaving an outage
indistinguishable from ordinary unauthenticated traffic.
Add a module-level logger to both web_ui handlers and log the
fetch failure with logger.exception() in the except block before
returning None. Behavior is unchanged (still fails closed); the
failure is now visible in CloudWatch. The secret value is never
logged. The two handlers stay byte-consistent in the mirrored
_get_auth_token() region.
The companion finding on the CDK import of the shared
procurement-ingest/web-ui-auth-token secret was evaluated and left
as-is: the token is a single secret shared by both the PO and WO
stacks, so from_secret_name_v2 (which scopes grant_read via the
standard 6-char suffix wildcard) is correct; making it a CDK-managed
Secret in both stacks would collide the two stacks on the same
explicit secret name at deploy time.
Refs: INFRA-74
* Make Cancelled PO status sticky via atomic write
The PO merge path read status with a get_item (_is_cancelled) and then
wrote with an unconditional update_item. Two defects followed from this:
- Race (Issue A): a cancellation landing between the read and the write
was silently un-cancelled by a revision carrying a non-cancelled
po_status — a TOCTOU on a table with concurrent email processing.
- Over-broad strip (Issue B): save_revision dropped po_status whenever
the PO was Cancelled, so legitimate status updates on non-cancelled
POs and status-less revisions were affected rather than only the true
un-cancel transition.
Enforce the invariant server-side instead. "Cancelled" is a sticky,
authoritative status: once set, later new_po/revision emails may enrich
other fields but must never move it to a non-cancelled status. When the
payload carries a non-cancelled po_status, _merge_update issues the
update_item guarded by ConditionExpression "attribute_not_exists(po_status)
OR po_status <> :marker", evaluated atomically at write time, so a
cancellation that lands first always wins. On ConditionalCheckFailedException
the same fields are re-written without po_status/cancelled_at, enriching the
record while Cancelled sticks. Payloads with no status change, or an already
-Cancelled status, take a plain merge — the status is only ever suppressed on
a real un-cancel. This removes the non-atomic get_item from the write path;
_is_cancelled is deleted. Key schema and attribute names are unchanged, so the
cross-stack purchase-orders contract (read-only by seahaven-slack-bot) holds.
Add moto-backed tests covering un-cancel suppression with field enrichment,
status-less merge onto a Cancelled PO, legitimate status updates on
non-cancelled POs, new_po backfill of a Cancelled skeleton, fresh
create/merge, and authoritative save_cancellation.
Refs: #97
2026-07-15 20:17:46 -04:00
|
|
|
web_ui_auth_secret.grant_read(web_ui)
|
2026-04-07 12:12:30 -04:00
|
|
|
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
# --- Throttles alarm: po-web-ui ---
|
|
|
|
|
web_ui.metric_throttles(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"WebUiThrottlesAlarm",
|
|
|
|
|
alarm_name="po-web-ui-throttles",
|
|
|
|
|
alarm_description="po-web-ui invocation throttles",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
|
|
|
|
# --- Duration alarm: po-web-ui ---
|
|
|
|
|
# Net-new (no orphan exists for this function).
|
|
|
|
|
# p99 / 45000 ms (75% of the 60s timeout) / eval 3, datapoints 2.
|
|
|
|
|
web_ui.metric_duration(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="p99",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"WebUiDurationAlarm",
|
|
|
|
|
alarm_name="po-web-ui-duration",
|
|
|
|
|
alarm_description="po-web-ui p99 duration approaching the 60s timeout",
|
|
|
|
|
threshold=45000,
|
|
|
|
|
evaluation_periods=3,
|
|
|
|
|
datapoints_to_alarm=2,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
Reconcile IaC with out-of-band DLQ + Function URL changes (INFRA-74, INFRA-41) (#50)
Make CDK the source of truth for two sets of changes applied out-of-band
via CLI to the po-ingest and WorkorderIngestStack stacks.
INFRA-74 (audit C-5): remove the public FunctionUrlAuthType.NONE Function
URL construct (and its auto-generated Principal:* invoke permission +
output) from both po-web-ui and workorder-web-ui. The URLs were already
deleted live via CLI; CFN's delete is idempotent.
INFRA-41 (audit H-8): add a CDK-managed SQS dead-letter queue
(dead_letter_queue=, 14d retention, SSL-enforced, CDK-generated name) and
an ALARM-only Errors alarm (Sum, threshold>0, site-alerts topic) for both
po-email-processor and workorder-email-processor, mirroring the
apm-wo-analysis-classifier DLQ and payments-payroll-batch alarm patterns.
Interim CLI resources (per-fn -dlq queues, -errors alarms, dlq-send inline
policies, OnFailure event-invoke-configs) removed post-deploy.
2026-06-08 16:02:29 -04:00
|
|
|
# Public Function URL removed 2026-06-08 (INFRA-74 / audit C-5): the
|
|
|
|
|
# unauthenticated FunctionUrlAuthType.NONE URL was deleted out-of-band
|
|
|
|
|
# via CLI. Removing the construct (and its auto-generated Principal:*
|
|
|
|
|
# invoke permission) reconciles IaC with the live state.
|
2026-04-30 14:26:53 -04:00
|
|
|
|
|
|
|
|
# --- Verified sites table (extracted from PO ship-to addresses) ---
|
|
|
|
|
verified_sites_table = dynamodb.Table(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"VerifiedSitesTable",
|
2026-04-30 14:26:53 -04:00
|
|
|
table_name="verified-sites",
|
|
|
|
|
partition_key=dynamodb.Attribute(
|
|
|
|
|
name="siteCode",
|
|
|
|
|
type=dynamodb.AttributeType.STRING,
|
|
|
|
|
),
|
|
|
|
|
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
|
|
|
|
|
removal_policy=RemovalPolicy.RETAIN,
|
|
|
|
|
)
|
2026-06-03 15:32:23 -04:00
|
|
|
# by-state GSI removed 2026-06-03 (audit M-20): 0 reads in 30d against
|
|
|
|
|
# 518 WCU of write amplification. Re-add if a state-level query path ships.
|
2026-04-30 14:26:53 -04:00
|
|
|
|
|
|
|
|
# --- Site extractor Lambda (DynamoDB Streams → verified-sites) ---
|
|
|
|
|
site_extractor = lambda_.Function(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"SiteExtractor",
|
2026-04-30 14:26:53 -04:00
|
|
|
function_name="po-ingest-site-extractor",
|
|
|
|
|
runtime=lambda_.Runtime.PYTHON_3_12,
|
|
|
|
|
architecture=lambda_.Architecture.ARM_64,
|
|
|
|
|
handler="handler.handler",
|
2026-05-12 15:21:06 -04:00
|
|
|
code=lambda_.Code.from_asset("../lambdas/po/site_extractor"),
|
2026-04-30 14:26:53 -04:00
|
|
|
timeout=Duration.seconds(60),
|
|
|
|
|
memory_size=256,
|
|
|
|
|
log_retention=logs.RetentionDays.TWO_MONTHS,
|
|
|
|
|
environment={
|
|
|
|
|
"VERIFIED_SITES_TABLE": verified_sites_table.table_name,
|
2026-04-30 15:01:09 -04:00
|
|
|
"PENDING_REVIEW_TABLE": "pending-site-review",
|
2026-04-30 14:26:53 -04:00
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
verified_sites_table.grant_read_write_data(site_extractor)
|
|
|
|
|
|
|
|
|
|
site_extractor.add_event_source(
|
|
|
|
|
lambda_event_sources.DynamoEventSource(
|
|
|
|
|
po_table,
|
|
|
|
|
starting_position=lambda_.StartingPosition.TRIM_HORIZON,
|
|
|
|
|
batch_size=10,
|
|
|
|
|
max_batching_window=Duration.seconds(30),
|
|
|
|
|
bisect_batch_on_error=True,
|
|
|
|
|
retry_attempts=3,
|
|
|
|
|
)
|
|
|
|
|
)
|
|
|
|
|
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
# --- Errors alarm: po-ingest-site-extractor ---
|
|
|
|
|
# Stream-consumer errors retry per the event-source config, but a
|
|
|
|
|
# persistent failure stalls the verified-sites pipeline. ALARM-only to
|
|
|
|
|
# site-alerts; no OK action; NOT_BREACHING when no data.
|
|
|
|
|
site_extractor.metric_errors(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"SiteExtractorErrorsAlarm",
|
|
|
|
|
alarm_name="po-ingest-site-extractor-errors",
|
|
|
|
|
alarm_description="po-ingest-site-extractor invocation errors",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
|
|
|
|
# --- Throttles alarm: po-ingest-site-extractor ---
|
|
|
|
|
site_extractor.metric_throttles(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="Sum",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"SiteExtractorThrottlesAlarm",
|
|
|
|
|
alarm_name="po-ingest-site-extractor-throttles",
|
|
|
|
|
alarm_description="po-ingest-site-extractor invocation throttles",
|
|
|
|
|
threshold=0,
|
|
|
|
|
evaluation_periods=1,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
|
|
|
|
# --- Duration alarm: po-ingest-site-extractor ---
|
|
|
|
|
# Net-new (no orphan exists for this function).
|
|
|
|
|
# p99 / 45000 ms (75% of the 60s timeout) / eval 3, datapoints 2.
|
|
|
|
|
site_extractor.metric_duration(
|
|
|
|
|
period=Duration.minutes(5),
|
|
|
|
|
statistic="p99",
|
|
|
|
|
).create_alarm(
|
|
|
|
|
self,
|
|
|
|
|
"SiteExtractorDurationAlarm",
|
|
|
|
|
alarm_name="po-ingest-site-extractor-duration",
|
|
|
|
|
alarm_description="po-ingest-site-extractor p99 duration approaching the 60s timeout",
|
|
|
|
|
threshold=45000,
|
|
|
|
|
evaluation_periods=3,
|
|
|
|
|
datapoints_to_alarm=2,
|
|
|
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
|
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
|
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
2026-05-08 16:01:21 -04:00
|
|
|
cdk.CfnOutput(
|
|
|
|
|
self,
|
|
|
|
|
"VerifiedSitesTableName",
|
2026-04-30 14:26:53 -04:00
|
|
|
value=verified_sites_table.table_name,
|
|
|
|
|
description="Verified site addresses extracted from POs",
|
|
|
|
|
)
|
2026-04-30 15:01:09 -04:00
|
|
|
|
|
|
|
|
# --- Pending site review table (POs with no extractable site code) ---
|
|
|
|
|
pending_review_table = dynamodb.Table(
|
2026-05-08 16:01:21 -04:00
|
|
|
self,
|
|
|
|
|
"PendingSiteReviewTable",
|
2026-04-30 15:01:09 -04:00
|
|
|
table_name="pending-site-review",
|
|
|
|
|
partition_key=dynamodb.Attribute(
|
|
|
|
|
name="po_number",
|
|
|
|
|
type=dynamodb.AttributeType.STRING,
|
|
|
|
|
),
|
|
|
|
|
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
|
|
|
|
|
removal_policy=RemovalPolicy.RETAIN,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
pending_review_table.grant_read_write_data(site_extractor)
|
|
|
|
|
verified_sites_table.grant_read_data(site_extractor)
|
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70)
* Add CloudWatch alarm coverage for po-ingest and workorder-ingest
Expands alarm coverage across both CDK stacks. All alarms are ALARM-only
(no OK action) to the shared site-alerts SNS topic, with TreatMissingData
NOT_BREACHING. The site-alerts topic is now imported once near the top of
each stack so every alarm reuses one Topic instance.
po-ingest (cdk/po_stack.py):
- Errors: po-ingest-site-extractor
- Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui
- Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption),
po-ingest-site-extractor, po-web-ui
- DynamoDB throttle + system-error: purchase-orders, verified-sites,
pending-site-review
workorder-ingest (cdk/wo_stack.py):
- Throttles: workorder-email-processor
- Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption)
- DynamoDB throttle + system-error: WorkOrders, WorkOrderComments
DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation
dimension set, so each table alarm is a Sum math expression across operations
via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests
is deprecated/invalid in aws-cdk-lib 2.259.0).
Refs INFRA-41 / audit H-8.
* Drop NEEDS ADAM SIGN-OFF wording from alarm comments
Duration alarm thresholds are owner-approved; remove the sign-off flag
from po_stack.py and wo_stack.py comments. Threshold values, eval config,
and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
|
|
|
|
|
|
|
|
# --- DynamoDB throttle + system-error alarms ---
|
|
|
|
|
# ThrottledRequests / SystemErrors emit at TableName + Operation only
|
|
|
|
|
# (verified against live CloudWatch: no TableName-only rollup exists, and
|
|
|
|
|
# metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0).
|
|
|
|
|
# Each table currently has zero throttle/error datapoints, so the series
|
|
|
|
|
# only materialise on first occurrence — NOT_BREACHING keeps them OK until
|
|
|
|
|
# then.
|
|
|
|
|
_add_ddb_alarms(
|
|
|
|
|
self, "PurchaseOrdersTable", po_table, "purchase-orders", alarm_topic
|
|
|
|
|
)
|
|
|
|
|
_add_ddb_alarms(
|
|
|
|
|
self,
|
|
|
|
|
"VerifiedSitesTable",
|
|
|
|
|
verified_sites_table,
|
|
|
|
|
"verified-sites",
|
|
|
|
|
alarm_topic,
|
|
|
|
|
)
|
|
|
|
|
_add_ddb_alarms(
|
|
|
|
|
self,
|
|
|
|
|
"PendingSiteReviewTable",
|
|
|
|
|
pending_review_table,
|
|
|
|
|
"pending-site-review",
|
|
|
|
|
alarm_topic,
|
|
|
|
|
)
|