Fix deploy-time failures found in prod testing

Two issues only a real deploy/run surfaced (synth + offline tests passed):

1. Classifier exceeded Lambda's 250 MB unzipped limit (bundled awswrangler +
   pandas + pyarrow + numpy). Move them to the AWS-managed SDK-for-pandas layer
   (AWSSDKPandas-Python312-Arm64:27, awswrangler 3.16.1, pre-stripped to fit);
   bundle only openpyxl. Drop the unused anthropic SDK — _call_haiku uses stdlib
   urllib. Function package now ~890 KB.

2. Slack rejected the daily post with invalid_blocks: every category drill
   button shared action_id "drill_category". Qualify it as "drill_category:<cat>"
   for uniqueness; the interactions handler now matches on the prefix. Add a
   regression test asserting all daily-summary action_ids are unique.

Verified in prod: classifier writes Parquet + summary.json + details.json;
slack-post posts the daily summary + 3rd-escalation alert; the interactions
endpoint (apm-wo.seahaven.com) returns 401 on a bad signature. 58/58 tests pass.

NOTE: these fixes sit on the phase-5 branch but logically belong to earlier
phases — the layer fix to #8 (classifier), the Slack fix to #10 — and must be
moved/cherry-picked there before those PRs merge independently. See cleanup.
This commit is contained in:
Adam Moussa 2026-05-28 18:29:16 -04:00
parent c04c239774
commit f193754c27
5 changed files with 42 additions and 5 deletions

View file

@ -64,6 +64,11 @@ from constructs import Construct
GLUE_DATABASE = "apm_wo_analysis"
GLUE_TABLE = "apm_wo_snapshots"
ATHENA_WORKGROUP = "apm-wo-analysis"
# AWS-managed SDK-for-pandas layer (awswrangler 3.16.1, py3.12, arm64). Provides
# awswrangler/pandas/pyarrow/numpy pre-stripped to fit the Lambda size limit.
AWSSDKPANDAS_LAYER_ARN = (
"arn:aws:lambda:us-east-1:336392948345:layer:AWSSDKPandas-Python312-Arm64:27"
)
ANTHROPIC_SECRET = "apm-wo-analysis/anthropic-api-key"
SLACK_SECRET = "apm-wo-analysis/slack-credentials"
GRAFANA_URL_PARAM = "/apm-wo-analysis/grafana-base-url"
@ -231,6 +236,15 @@ class PipelineStack(Stack):
timeout=Duration.seconds(120),
log_group=classifier_logs,
environment={"APM_HAIKU_FALLBACK": "on"},
# awswrangler/pandas/pyarrow/numpy come from the AWS-managed
# SDK-for-pandas layer (pre-stripped to fit the 250 MB unzipped
# limit, which bundling them ourselves blows). The function package
# only bundles openpyxl; boto3 is in the runtime, urllib is stdlib.
layers=[
lambda_.LayerVersion.from_layer_version_arn(
self, "PandasLayer", AWSSDKPANDAS_LAYER_ARN
)
],
code=lambda_.Code.from_asset(
os.path.join(LAMBDAS_DIR, "classifier"),
bundling=BundlingOptions(

View file

@ -1,3 +1,5 @@
awswrangler>=3.9.0
# awswrangler + pandas/pyarrow/numpy come from the AWS-managed SDK-for-pandas
# Lambda layer (see pipeline_stack.py) — bundling them here blows the 250 MB
# unzipped limit. The Haiku fallback uses stdlib urllib, so no anthropic SDK.
# Only openpyxl (xlsx parsing) is bundled into the function package.
openpyxl>=3.1.0
anthropic>=0.40.0

View file

@ -282,8 +282,11 @@ def build_daily_summary(
for cat, count in drill_cats:
short_label = cat.replace(" Escalation", " Esc.").replace("Awaiting ", "")
short_label = short_label[:20] # button text kept concise
# action_id must be UNIQUE per message (Slack rejects duplicates), so
# qualify it with the category; the interactions handler matches on the
# "drill_category:" prefix and reads the filter value from `value`.
button_elements.append(
_button(f"{short_label} ({count})", "drill_category", cat)
_button(f"{short_label} ({count})", f"drill_category:{cat}", cat)
)
# Dashboard link button always present (no action_id — url button).

View file

@ -49,7 +49,10 @@ def handler(event, context):
return {"statusCode": 200, "body": ""}
action = (payload.get("actions") or [{}])[0]
field = _FILTER_FIELD.get(action.get("action_id"))
# action_id is qualified for Slack uniqueness, e.g. "drill_category:Report /
# Docs Needed" — match on the prefix before ":"; the filter value is in `value`.
action_kind = (action.get("action_id") or "").split(":", 1)[0]
field = _FILTER_FIELD.get(action_kind)
value = action.get("value")
trigger_id = payload.get("trigger_id")
if not field or not value or not trigger_id:

View file

@ -151,6 +151,20 @@ class TestDailySummaryStructure:
blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL)
assert any(b.get("type") == "header" for b in blocks)
def test_action_ids_are_unique(self):
# Slack rejects a message with duplicate action_ids across its elements
# (regression: all drill buttons once shared action_id "drill_category").
blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL)
action_ids = [
el["action_id"]
for b in blocks
for el in b.get("elements", [])
if isinstance(el, dict) and "action_id" in el
]
assert len(action_ids) == len(set(action_ids)), (
f"duplicate action_id(s): {action_ids}"
)
def test_header_contains_date(self):
blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL)
header = next(b for b in blocks if b.get("type") == "header")
@ -191,7 +205,8 @@ class TestDailySummaryStructure:
if block.get("type") != "actions":
continue
for elem in block.get("elements", []):
if elem.get("action_id") == "drill_category":
# action_id is qualified for uniqueness: "drill_category:<cat>".
if str(elem.get("action_id", "")).startswith("drill_category:"):
drill_found = True
assert "value" in elem
assert drill_found, "expected at least one drill_category action button"