mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-09-30 06:03:14 +00:00
Some checks failed
Deploy / deploy (push) Has been cancelled
Both email-processor Code.from_asset calls now bundle from lambdas/ instead of their per-function subdirectory, so Phase 3's shared/ module is reachable from the asset root once it lands. The bundling commands were rewritten for the new cwd (pip install -r <po|wo>/ email_processor/requirements.txt -t /asset-output && cp <po|wo>/ email_processor/*.py /asset-output/), preserving the ARM64 --platform manylinux2014_aarch64 --only-binary=:all: pin exactly — its removal shipped x86 wheels into the ARM64 function and caused a 100% outage (PR #34). All five from_asset calls (both email processors, po web_ui, po site_extractor, wo web_ui) now exclude **/__pycache__/**; the two widened ones also exclude **/tests/** and **/package/**. Without the package/ exclude, the stale untracked 44 MB lambdas/po/email_processor/package/ dir (local-only, never present in CI) would diverge local vs CI asset hashes and force spurious redeploys — from_asset doesn't honor .gitignore. That dir is left in place; deleting it is Adam's call. WO's prod zip shrinks as deliberate cleanup, not a byte-identical match to PO: the old `cp -r .` shipped tests/ (real scrubbed .eml fixtures), __pycache__/, and requirements.txt into production. The acceptance bar for WO is runtime-imported module set unchanged + smoke, not a byte-identical zip; PO keeps the byte-identical first-party file set guarantee. tests/test_bundle_consistency.py is updated in the same change to recognize the scoped `cp po/email_processor/*.py` (resp. wo) glob as the new unconditionally-safe shape, without loosening the allowlist-revert detection, the detection-logic mutation test, or the PO_EXPECTED_TOP_LEVEL_MODULES exact-set pin. No code moved under lambdas/ in this change (git diff main...HEAD -- lambdas/ is empty); only CDK asset wiring and its tests changed.
588 lines
28 KiB
Python
588 lines
28 KiB
Python
"""CDK stack for the work order email ingestion pipeline."""
|
|
|
|
import aws_cdk as cdk
|
|
from aws_cdk import (
|
|
Duration,
|
|
RemovalPolicy,
|
|
Stack,
|
|
aws_cloudwatch as cloudwatch,
|
|
aws_cloudwatch_actions as cw_actions,
|
|
aws_dynamodb as dynamodb,
|
|
aws_iam as iam,
|
|
aws_lambda as lambda_,
|
|
aws_logs as logs,
|
|
aws_s3 as s3,
|
|
aws_s3_notifications as s3n,
|
|
aws_ses as ses,
|
|
aws_ses_actions as ses_actions,
|
|
aws_secretsmanager as secretsmanager,
|
|
aws_sns as sns,
|
|
aws_sqs as sqs,
|
|
)
|
|
from constructs import Construct
|
|
|
|
# Operations these tables actually issue (PutItem/UpdateItem/DeleteItem writes,
|
|
# GetItem/Query/BatchGetItem reads). DynamoDB emits ThrottledRequests/SystemErrors
|
|
# keyed by TableName + Operation only, so the CDK *_for_operations helpers (which
|
|
# render a SUM MathExpression across these per-operation metrics) are the correct,
|
|
# non-deprecated way to roll a table up to a single alarmable series.
|
|
_DDB_ALARM_OPERATIONS = [
|
|
dynamodb.Operation.GET_ITEM,
|
|
dynamodb.Operation.BATCH_GET_ITEM,
|
|
dynamodb.Operation.QUERY,
|
|
dynamodb.Operation.SCAN,
|
|
dynamodb.Operation.PUT_ITEM,
|
|
dynamodb.Operation.UPDATE_ITEM,
|
|
dynamodb.Operation.DELETE_ITEM,
|
|
dynamodb.Operation.BATCH_WRITE_ITEM,
|
|
]
|
|
|
|
|
|
def _add_ddb_alarms(scope, id_prefix, table, alarm_name_prefix, alarm_topic):
|
|
"""Add throttle + system-error alarms for a DynamoDB table.
|
|
|
|
Both fire on any non-zero datapoint in a 5-min window. ALARM-only SnsAction
|
|
to site-alerts (no OK action); TreatMissingData NOT_BREACHING.
|
|
"""
|
|
table.metric_throttled_requests_for_operations(
|
|
operations=_DDB_ALARM_OPERATIONS,
|
|
period=Duration.minutes(5),
|
|
statistic="Sum",
|
|
).create_alarm(
|
|
scope,
|
|
f"{id_prefix}ThrottlesAlarm",
|
|
alarm_name=f"{alarm_name_prefix}-throttles",
|
|
alarm_description=f"{alarm_name_prefix} DynamoDB throttled requests",
|
|
threshold=0,
|
|
evaluation_periods=1,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
table.metric_system_errors_for_operations(
|
|
operations=_DDB_ALARM_OPERATIONS,
|
|
period=Duration.minutes(5),
|
|
statistic="Sum",
|
|
).create_alarm(
|
|
scope,
|
|
f"{id_prefix}SystemErrorsAlarm",
|
|
alarm_name=f"{alarm_name_prefix}-system-errors",
|
|
alarm_description=f"{alarm_name_prefix} DynamoDB server-side (5xx) errors",
|
|
threshold=0,
|
|
evaluation_periods=1,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
# CloudWatch namespace for the log-derived sender-authentication metrics.
|
|
_SENDER_AUTH_METRIC_NAMESPACE = "Seahaven/ProcurementIngest"
|
|
|
|
|
|
def _add_sender_auth_rejected_alarm(scope, id_prefix, function_name, alarm_topic):
|
|
"""Metric-filter + alarm on ``sender_auth_rejected`` warnings (INFRA-107).
|
|
|
|
A rejected inbound email is skipped without erroring the invocation, so it
|
|
is invisible to the Errors/Throttles/DLQ alarms. This turns the structured
|
|
warning log into a CloudWatch metric and pages when rejections spike --
|
|
catching a silent false-reject storm (allowlist wrong, signing-domain
|
|
drift, SES header-format change) that would otherwise discard legitimate
|
|
mail while the pipeline reports healthy. This is the safety net for the WO
|
|
allowlist domain assumption (seahaven.com) -- if the real Gmail-forward
|
|
re-signing domain differs, this alarm surfaces it instead of a silent
|
|
work-order outage.
|
|
|
|
ALARM-only SnsAction to site-alerts; no OK action. The metric filter reads
|
|
the function's own log group (imported by the deterministic
|
|
``/aws/lambda/<fn>`` name, created by the function's log_retention). A plain
|
|
substring pattern is used because Lambda prefixes each line with its own
|
|
level/timestamp/request-id, so the JSON payload is not a standalone JSON
|
|
log event a `{$.event=...}` pattern could match.
|
|
"""
|
|
metric_name = f"{function_name}-sender-auth-rejected"
|
|
logs.MetricFilter(
|
|
scope,
|
|
f"{id_prefix}SenderAuthRejectedFilter",
|
|
log_group=logs.LogGroup.from_log_group_name(
|
|
scope,
|
|
f"{id_prefix}LogGroup",
|
|
f"/aws/lambda/{function_name}",
|
|
),
|
|
filter_pattern=logs.FilterPattern.literal('"sender_auth_rejected"'),
|
|
metric_namespace=_SENDER_AUTH_METRIC_NAMESPACE,
|
|
metric_name=metric_name,
|
|
metric_value="1",
|
|
default_value=0,
|
|
)
|
|
|
|
# Fire on a *sustained* reject condition rather than a volume spike. The
|
|
# earlier Sum>=3-over-15-min threshold had a blind spot that is exactly the
|
|
# failure this alarm exists to catch: a low-traffic pipeline in total
|
|
# drift outage (allowlist wrong / signing-domain changed) may only produce
|
|
# a trickle of rejects -- one every few minutes -- that never sums to 3 in
|
|
# any window, so the outage never pages. Instead: >=1 reject per 5-min
|
|
# period, alarming when 2 of the last 6 periods breach (evaluation_periods=6
|
|
# / datapoints_to_alarm=2). Six periods (30 min) with only 2 required
|
|
# datapoints closes the sparse-outage residual: even rejections >10-15 min
|
|
# apart can still place two breaching datapoints in a single 30-min
|
|
# evaluation window. A single stray spoof probe (one lone period) is
|
|
# tolerated and self-clears, but a sustained reject condition trips even
|
|
# at very low arrival rates. default_value=0 on the metric filter keeps the
|
|
# series continuous so NOT_BREACHING only applies before the first datapoint
|
|
# ever arrives.
|
|
cloudwatch.Metric(
|
|
namespace=_SENDER_AUTH_METRIC_NAMESPACE,
|
|
metric_name=metric_name,
|
|
period=Duration.minutes(5),
|
|
statistic="Sum",
|
|
).create_alarm(
|
|
scope,
|
|
f"{id_prefix}SenderAuthRejectedAlarm",
|
|
alarm_name=f"{function_name}-sender-auth-rejected",
|
|
alarm_description=(
|
|
f"{function_name} rejected inbound mail on sender authentication "
|
|
"(possible allowlist/DKIM-domain drift silently dropping real mail)"
|
|
),
|
|
threshold=1,
|
|
evaluation_periods=6,
|
|
datapoints_to_alarm=2,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
|
|
class WorkorderIngestStack(Stack):
|
|
def __init__(self, scope: Construct, construct_id: str, **kwargs):
|
|
super().__init__(scope, construct_id, **kwargs)
|
|
|
|
# --- Shared alarm SNS topic (site-alerts) ---
|
|
# Imported once near the top so every alarm in this stack reuses the same
|
|
# Topic construct instance (avoids duplicate logical IDs). ALARM-only
|
|
# SnsAction; no OK action, per the CloudWatch-alarm preference. The
|
|
# topic's CMK (alias/seahaven-alarm-topics) lives on the topic itself.
|
|
alarm_topic = sns.Topic.from_topic_arn(
|
|
self,
|
|
"SiteAlertsTopic",
|
|
f"arn:aws:sns:{self.region}:{self.account}:site-alerts",
|
|
)
|
|
|
|
# --- S3 bucket for raw emails ---
|
|
email_bucket = s3.Bucket(
|
|
self,
|
|
"EmailBucket",
|
|
bucket_name=f"workorder-ingest-emails-{self.account}",
|
|
block_public_access=s3.BlockPublicAccess.BLOCK_ALL,
|
|
removal_policy=RemovalPolicy.RETAIN,
|
|
lifecycle_rules=[
|
|
s3.LifecycleRule(expiration=Duration.days(90)),
|
|
],
|
|
)
|
|
|
|
# --- DynamoDB tables ---
|
|
work_orders_table = dynamodb.Table(
|
|
self,
|
|
"WorkOrdersTable",
|
|
table_name="WorkOrders",
|
|
partition_key=dynamodb.Attribute(
|
|
name="work_order_id",
|
|
type=dynamodb.AttributeType.STRING,
|
|
),
|
|
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
|
|
removal_policy=RemovalPolicy.RETAIN,
|
|
)
|
|
# site-code-index and status-index GSIs removed 2026-06-03 (audit M-20):
|
|
# 0 reads in 30d against ~50k WCU each of write amplification. Re-add if
|
|
# a site-code or status query path ships.
|
|
|
|
comments_table = dynamodb.Table(
|
|
self,
|
|
"CommentsTable",
|
|
table_name="WorkOrderComments",
|
|
partition_key=dynamodb.Attribute(
|
|
name="work_order_id",
|
|
type=dynamodb.AttributeType.STRING,
|
|
),
|
|
sort_key=dynamodb.Attribute(
|
|
name="comment_id",
|
|
type=dynamodb.AttributeType.STRING,
|
|
),
|
|
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
|
|
removal_policy=RemovalPolicy.RETAIN,
|
|
)
|
|
|
|
# --- Anthropic API key secret removed (Bedrock migration) ---
|
|
# Parsing moved from the Anthropic API to the Bedrock inference profile
|
|
# us.anthropic.claude-haiku-4-5-20251001-v1:0, so no provider API key is
|
|
# needed. The old secret "workorder-ingest/anthropic-api-key" had
|
|
# RemovalPolicy.RETAIN, so it is ORPHANED (not deleted) by this change:
|
|
# delete it manually post-deploy and revoke the stored key at Anthropic.
|
|
|
|
# --- DLQ for failed async invocations (INFRA-41 / audit H-8) ---
|
|
# SES → S3 → Lambda is async; without an OnFailure destination a failed
|
|
# parse (bad email, transient error) is silently dropped after Lambda's
|
|
# retries. CDK generates the queue name to avoid colliding with the
|
|
# interim CLI-created workorder-email-processor-dlq (removed post-deploy).
|
|
email_processor_dlq = sqs.Queue(
|
|
self,
|
|
"EmailProcessorDlq",
|
|
retention_period=Duration.days(14),
|
|
enforce_ssl=True,
|
|
)
|
|
|
|
# --- Lambda function ---
|
|
email_processor = lambda_.Function(
|
|
self,
|
|
"EmailProcessor",
|
|
function_name="workorder-email-processor",
|
|
runtime=lambda_.Runtime.PYTHON_3_12,
|
|
architecture=lambda_.Architecture.ARM_64,
|
|
handler="handler.handler",
|
|
code=lambda_.Code.from_asset(
|
|
"../lambdas",
|
|
exclude=["**/__pycache__/**", "**/tests/**", "**/package/**"],
|
|
bundling=cdk.BundlingOptions(
|
|
image=lambda_.Runtime.PYTHON_3_12.bundling_image,
|
|
command=[
|
|
"bash",
|
|
"-c",
|
|
"pip install --platform manylinux2014_aarch64 --only-binary=:all: "
|
|
"-r wo/email_processor/requirements.txt -t /asset-output && "
|
|
"cp wo/email_processor/*.py /asset-output/",
|
|
],
|
|
),
|
|
),
|
|
timeout=Duration.seconds(60),
|
|
memory_size=256,
|
|
log_retention=logs.RetentionDays.TWO_MONTHS,
|
|
dead_letter_queue=email_processor_dlq,
|
|
environment={
|
|
"WORK_ORDERS_TABLE": work_orders_table.table_name,
|
|
"COMMENTS_TABLE": comments_table.table_name,
|
|
"BEDROCK_MODEL_ID": "us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
# Fail-closed sender auth (INFRA-107): the handler only
|
|
# accepts mail whose SES-stamped Authentication-Results
|
|
# header carries dkim=pass for one of these domains. APM
|
|
# mail arrives via the apm@ Google Groups forward, which
|
|
# re-signs as seahaven.com (observed on live traffic
|
|
# 2026-07-15: "dkim=pass header.i=@seahaven.com"; the
|
|
# original hxgnsmartcloud.com signature does not survive
|
|
# the forward). Unset/empty ⇒ the handler rejects all mail.
|
|
"ALLOWED_DKIM_DOMAINS": "seahaven.com",
|
|
},
|
|
)
|
|
|
|
# Grant permissions
|
|
email_bucket.grant_read(email_processor)
|
|
work_orders_table.grant_read_write_data(email_processor)
|
|
comments_table.grant_read_write_data(email_processor)
|
|
|
|
# --- Bedrock InvokeModel grant ---
|
|
# The us.* inference profile can route cross-region, so the grant MUST
|
|
# cover both the inference-profile ARN AND the per-region foundation-model
|
|
# ARNs (empty account field) for every region the profile can reach
|
|
# (us-east-1/us-east-2/us-west-2). A profile-only grant AccessDenies at
|
|
# runtime whenever the profile routes to a region whose foundation-model
|
|
# ARN is not allowed.
|
|
email_processor.add_to_role_policy(
|
|
iam.PolicyStatement(
|
|
actions=[
|
|
"bedrock:InvokeModel",
|
|
"bedrock:InvokeModelWithResponseStream",
|
|
],
|
|
resources=[
|
|
"arn:aws:bedrock:us-east-1:328440206208:inference-profile/us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
"arn:aws:bedrock:us-east-1::foundation-model/anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
"arn:aws:bedrock:us-east-2::foundation-model/anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
"arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
],
|
|
)
|
|
)
|
|
|
|
# NOTE: The pre-emptive grant_encrypt_decrypt on the shared DynamoDB CMK
|
|
# (alias/seahaven-dynamodb) was removed (security sweep 2026-06-17). The
|
|
# WorkOrders/WorkOrderComments tables are NOT SSE-KMS encrypted with that
|
|
# CMK, so the grant was unused for these tables yet handed
|
|
# wo-email-processor kms:Decrypt on the CMK that also protects the
|
|
# purchase-orders table (cross-stack decrypt reach). Re-add this grant only
|
|
# as part of the actual CMK migration of these tables (INFRA-6), at which
|
|
# point grant_read_write_data on the (then encrypted) tables would propagate
|
|
# the needed key permissions automatically.
|
|
|
|
# --- Errors alarm (INFRA-41 / audit H-8) ---
|
|
# ALARM-only (no OK action, per the CloudWatch-alarm preference) to the
|
|
# shared site-alerts topic. Any errored invocation in a 5-min window pages.
|
|
email_processor.metric_errors(
|
|
period=Duration.minutes(5),
|
|
statistic="Sum",
|
|
).create_alarm(
|
|
self,
|
|
"EmailProcessorErrorsAlarm",
|
|
alarm_name="workorder-email-processor-errors",
|
|
alarm_description="workorder-email-processor async invocation errors",
|
|
threshold=0,
|
|
evaluation_periods=1,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
# --- Sender-auth rejection alarm (INFRA-107) ---
|
|
# A rejected email (bad/unaligned DKIM verdict) returns normally, so it
|
|
# produces NO Lambda error, NO DLQ message and NO retry -- only a
|
|
# `sender_auth_rejected` warning log. The WO allowlist trusts dkim=pass
|
|
# for seahaven.com on the assumption the apm@ forward re-signs there; if
|
|
# that assumption is wrong (e.g. a Gmail auto-forward re-signs under a
|
|
# different domain), 100% of legitimate work-order mail is silently
|
|
# dropped. This metric filter + alarm turns those warnings into a paging
|
|
# signal so a false-reject storm surfaces instead of a silent outage.
|
|
_add_sender_auth_rejected_alarm(
|
|
self, "EmailProcessor", "workorder-email-processor", alarm_topic
|
|
)
|
|
|
|
# --- Throttles alarm: workorder-email-processor ---
|
|
# Any throttled invocation (concurrency cap hit) in a 5-min window pages.
|
|
# ALARM-only to site-alerts; no OK action; NOT_BREACHING when no data.
|
|
email_processor.metric_throttles(
|
|
period=Duration.minutes(5),
|
|
statistic="Sum",
|
|
).create_alarm(
|
|
self,
|
|
"EmailProcessorThrottlesAlarm",
|
|
alarm_name="workorder-email-processor-throttles",
|
|
alarm_description="workorder-email-processor invocation throttles",
|
|
threshold=0,
|
|
evaluation_periods=1,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
# --- Duration alarm: workorder-email-processor (orphan adoption) ---
|
|
# Adopts the orphaned CLI alarm Lambda-Duration-workorder-email-processor
|
|
# under the repo's <fn>-duration naming (NEW logical name → no deploy
|
|
# collision; delete the orphan post-deploy). p95 /
|
|
# 45000 ms (75% of the 60s timeout) / eval 3 of which 2 datapoints —
|
|
# tighter than the orphan's Maximum>=48000 / 1-of-1.
|
|
email_processor.metric_duration(
|
|
period=Duration.minutes(5),
|
|
statistic="p95",
|
|
).create_alarm(
|
|
self,
|
|
"EmailProcessorDurationAlarm",
|
|
alarm_name="workorder-email-processor-duration",
|
|
alarm_description="workorder-email-processor p95 duration approaching the 60s timeout",
|
|
threshold=45000,
|
|
evaluation_periods=3,
|
|
datapoints_to_alarm=2,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
# --- DLQ messages-present alarm (INFRA-41 / audit H-8) ---
|
|
# The errors alarm above fires on any errored invocation, but a message
|
|
# only lands in the DLQ after Lambda exhausts its async retries and gives
|
|
# up — i.e. a genuinely dropped email. ALARM-only (no OK action) to the
|
|
# same shared site-alerts topic. MAXIMUM over a 5-min window so a single
|
|
# visible message pages even if it is later consumed/redriven.
|
|
email_processor_dlq.metric_approximate_number_of_messages_visible(
|
|
period=Duration.minutes(5),
|
|
statistic="Maximum",
|
|
).create_alarm(
|
|
self,
|
|
"EmailProcessorDlqMessagesAlarm",
|
|
alarm_name="workorder-email-processor-dlq-messages",
|
|
alarm_description="workorder-email-processor DLQ has visible messages (dropped emails)",
|
|
threshold=0,
|
|
evaluation_periods=1,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
# --- Template fallback-rate alarm: workorder-email-processor ---
|
|
# The processor tries a deterministic template parse first and only calls
|
|
# the Bedrock AI extractor on a miss/invalid. A sustained rise in the
|
|
# ai_fallback share signals Hexagon template drift (coverage collapse).
|
|
# EMF metric Seahaven/WorkorderIngest/ParseOutcome, dimensioned by
|
|
# ParseMethod (template|ai_fallback). 15-min periods (deliberate deviation
|
|
# from the 5-min house style) accumulate a stable denominator at the low
|
|
# ~760/day volume; FILL(0) + a >=10-sample volume floor prevent
|
|
# low-volume false pages and INSUFFICIENT_DATA. ALARM-only SnsAction to
|
|
# site-alerts, no OK action, NOT_BREACHING -- matching the stack idiom.
|
|
fb_metric = cloudwatch.Metric(
|
|
namespace="Seahaven/WorkorderIngest",
|
|
metric_name="ParseOutcome",
|
|
dimensions_map={"ParseMethod": "ai_fallback"},
|
|
statistic="Sum",
|
|
period=Duration.minutes(15),
|
|
)
|
|
# AI-fallback parses REJECTED by the validate_ai_fallback gate emit
|
|
# ParseMethod=ai_fallback_rejected (and nothing else), so they must
|
|
# count as fallback here too -- otherwise a drift outage whose AI
|
|
# output also fails the gate would LOWER the observed fallback rate
|
|
# while silently dropping mail.
|
|
fb_rej_metric = cloudwatch.Metric(
|
|
namespace="Seahaven/WorkorderIngest",
|
|
metric_name="ParseOutcome",
|
|
dimensions_map={"ParseMethod": "ai_fallback_rejected"},
|
|
statistic="Sum",
|
|
period=Duration.minutes(15),
|
|
)
|
|
tmpl_metric = cloudwatch.Metric(
|
|
namespace="Seahaven/WorkorderIngest",
|
|
metric_name="ParseOutcome",
|
|
dimensions_map={"ParseMethod": "template"},
|
|
statistic="Sum",
|
|
period=Duration.minutes(15),
|
|
)
|
|
fallback_rate = cloudwatch.MathExpression(
|
|
expression=(
|
|
# The IF volume floor (>=10) already guarantees the denominator
|
|
# is non-zero in the true branch, so divide directly. (An earlier
|
|
# MAX([...,1]) divide-by-zero guard used array syntax CloudWatch
|
|
# rejects at deploy: "Unsupported operand type(s) for MAX".)
|
|
"IF((FILL(fb,0)+FILL(rej,0)+FILL(tmpl,0))>=10, "
|
|
"100*(FILL(fb,0)+FILL(rej,0))"
|
|
"/(FILL(fb,0)+FILL(rej,0)+FILL(tmpl,0)), 0)"
|
|
),
|
|
using_metrics={
|
|
"fb": fb_metric,
|
|
"rej": fb_rej_metric,
|
|
"tmpl": tmpl_metric,
|
|
},
|
|
period=Duration.minutes(15),
|
|
label="TemplateFallbackRatePct",
|
|
)
|
|
fallback_rate.create_alarm(
|
|
self,
|
|
"EmailProcessorTemplateFallbackRateAlarm",
|
|
alarm_name="workorder-email-processor-template-fallback-rate",
|
|
alarm_description=(
|
|
"workorder-email-processor deterministic-template coverage "
|
|
"collapse: >15% of parses fell back to the Bedrock AI extractor"
|
|
),
|
|
threshold=15,
|
|
evaluation_periods=3,
|
|
datapoints_to_alarm=2,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
# --- AI-fallback rejected alarm: workorder-email-processor ---
|
|
# A parse rejected by the validate_ai_fallback gate is dropped without
|
|
# error/retry/DLQ (fail closed), so like sender-auth rejections it
|
|
# needs its own pager or a sustained rejection condition (prompt-
|
|
# injection probing, or template drift whose AI output fails the gate)
|
|
# stays silent. Same sparse-arrival idiom as the sender-auth-rejected
|
|
# alarm: >=1 rejection per 5-min period, 2 of the last 6 periods (30
|
|
# min), so a lone probe self-clears but a burst pages within ~10 min.
|
|
# Coverage residual (matching the sender-auth-rejected sibling and
|
|
# knowingly accepted): rejections spaced >~25-30 min apart never place
|
|
# two breaching datapoints in one 30-min window, and the fallback-rate
|
|
# alarm dilutes them below 15% against normal template volume, so a
|
|
# *very* sparse silent-drop trickle is not paged by either alarm.
|
|
# EMF emits no datapoint in quiet periods (no metric-filter
|
|
# default_value here); NOT_BREACHING treats those gaps as OK.
|
|
cloudwatch.Metric(
|
|
namespace="Seahaven/WorkorderIngest",
|
|
metric_name="ParseOutcome",
|
|
dimensions_map={"ParseMethod": "ai_fallback_rejected"},
|
|
statistic="Sum",
|
|
period=Duration.minutes(5),
|
|
).create_alarm(
|
|
self,
|
|
"EmailProcessorAiFallbackRejectedAlarm",
|
|
alarm_name="workorder-email-processor-ai-fallback-rejected",
|
|
alarm_description=(
|
|
"workorder-email-processor is rejecting Bedrock AI-fallback "
|
|
"output at the validation gate (possible prompt-injection "
|
|
"probing or template drift silently dropping real mail)"
|
|
),
|
|
threshold=1,
|
|
evaluation_periods=6,
|
|
datapoints_to_alarm=2,
|
|
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
|
|
|
|
# S3 event notification -> Lambda
|
|
email_bucket.add_event_notification(
|
|
s3.EventType.OBJECT_CREATED,
|
|
s3n.LambdaDestination(email_processor),
|
|
s3.NotificationKeyFilter(prefix="inbound/"),
|
|
)
|
|
|
|
# --- SES Receipt Rule ---
|
|
rule_set = ses.ReceiptRuleSet.from_receipt_rule_set_name(
|
|
self,
|
|
"ExistingRuleSet",
|
|
"INBOUND_MAIL",
|
|
)
|
|
|
|
rule_set.add_rule(
|
|
"WorkorderEmailRule",
|
|
recipients=["apm@int.seahaven.com"],
|
|
actions=[
|
|
ses_actions.S3(
|
|
bucket=email_bucket,
|
|
object_key_prefix="inbound/",
|
|
),
|
|
],
|
|
)
|
|
|
|
# --- Web UI auth token secret ---
|
|
# Shared secret for the web UI auth gate, stored in Secrets Manager and
|
|
# resolved at runtime so the token never appears in CloudFormation templates
|
|
# or Lambda environment variables. Create this secret before deploying
|
|
# either stack; both PO and WO stacks reference it by name.
|
|
web_ui_auth_secret = secretsmanager.Secret.from_secret_name_v2(
|
|
self,
|
|
"WebUiAuthToken",
|
|
"procurement-ingest/web-ui-auth-token",
|
|
)
|
|
|
|
# --- Web UI Lambda ---
|
|
web_ui = lambda_.Function(
|
|
self,
|
|
"WebUI",
|
|
function_name="workorder-web-ui",
|
|
runtime=lambda_.Runtime.PYTHON_3_12,
|
|
architecture=lambda_.Architecture.ARM_64,
|
|
handler="handler.handler",
|
|
code=lambda_.Code.from_asset(
|
|
"../lambdas/wo/web_ui", exclude=["**/__pycache__/**"]
|
|
),
|
|
timeout=Duration.seconds(15),
|
|
memory_size=128,
|
|
log_retention=logs.RetentionDays.TWO_MONTHS,
|
|
environment={
|
|
"WORK_ORDERS_TABLE": work_orders_table.table_name,
|
|
"COMMENTS_TABLE": comments_table.table_name,
|
|
# Defense-in-depth shared secret for the web UI handler. The
|
|
# handler fails closed if this ARN is unset or the secret is
|
|
# missing, so any future invocation path cannot re-expose the
|
|
# WO DB unauthenticated. The secret value is fetched at runtime
|
|
# from Secrets Manager (not embedded in env vars or template).
|
|
"WEB_UI_AUTH_TOKEN_SECRET_ARN": web_ui_auth_secret.secret_arn,
|
|
},
|
|
)
|
|
|
|
work_orders_table.grant_read_data(web_ui)
|
|
comments_table.grant_read_data(web_ui)
|
|
web_ui_auth_secret.grant_read(web_ui)
|
|
|
|
# --- DynamoDB throttle + system-error alarms ---
|
|
# ThrottledRequests / SystemErrors emit at TableName + Operation only
|
|
# (verified against live CloudWatch: no TableName-only rollup exists, and
|
|
# metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0).
|
|
# Each table currently has zero throttle/error datapoints, so the series
|
|
# only materialise on first occurrence — NOT_BREACHING keeps them OK until
|
|
# then.
|
|
_add_ddb_alarms(
|
|
self, "WorkOrdersTable", work_orders_table, "WorkOrders", alarm_topic
|
|
)
|
|
_add_ddb_alarms(
|
|
self, "WorkOrderComments", comments_table, "WorkOrderComments", alarm_topic
|
|
)
|
|
|
|
# Public Function URL removed 2026-06-08 (INFRA-74 / audit C-5): the
|
|
# unauthenticated FunctionUrlAuthType.NONE URL was deleted out-of-band
|
|
# via CLI. Removing the construct (and its auto-generated Principal:*
|
|
# invoke permission) reconciles IaC with the live state.
|