From fa2397cb5dfeb74ea5fd51c6e9c2e9eead3d4c51 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Mon, 11 May 2026 16:56:25 -0400 Subject: [PATCH] Add business hours gating, start date, and morning summary Restrict alerts to 8am-5pm ET Mon-Fri. Lambda skips before START_DATE (2026-05-14). First run of day with 2+ breaches sends consolidated summary to #front-sla-alerts. Cron narrowed to UTC 12-22 for DST coverage. README updated with new config and architecture. --- README.md | 24 ++++++--- src/monitor/app.py | 120 +++++++++++++++++++++++++++++++++++++-------- template.yaml | 9 +++- 3 files changed, 122 insertions(+), 31 deletions(-) diff --git a/README.md b/README.md index bdfbeeb..79d0e08 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # front-sla-monitor -Scheduled Lambda that monitors Front conversations for SLA breaches and sends tiered Slack alerts. Runs every 15 minutes on weekdays. +Scheduled Lambda that monitors Front conversations for SLA breaches and sends tiered Slack alerts. Runs every 15 minutes during business hours (8 AM–5 PM ET, Mon-Fri). ## SLA Rules @@ -11,22 +11,28 @@ Scheduled Lambda that monitors Front conversations for SLA breaches and sends ti Business time counts weekday hours only (Mon-Fri, Eastern time). The SLA clock pauses on Saturday and Sunday. +Alerts are only sent during business hours (8 AM–5 PM ET). If breaches accumulate overnight, the first morning run sends a single summary message to #front-sla-alerts instead of individual alerts. + ## Architecture ``` -EventBridge (every 15 min, Mon-Fri) +EventBridge (every 15 min, 8 AM–5 PM ET, Mon-Fri) │ ▼ Lambda (Python 3.12, arm64) + │ + ├── Start date gate → skip before START_DATE + ├── Business hours gate → skip outside 8 AM–5 PM ET │ ├── Secrets Manager → front-sla-monitor/front-api-token ├── Secrets Manager → front-sla-monitor/slack-bot-token │ - ├── GET Front API /inboxes → list shared inboxes - ├── GET Front API /inboxes/{id}/conversations → open conversations + ├── GET Front API /inboxes → filter to MONITOR_INBOXES + ├── GET Front API /inboxes/{id}/conversations → open conversations (max 10 pages) │ - ├── DynamoDB (front-sla-alerts) → dedup: skip already-alerted conversations + ├── DynamoDB (front-sla-alerts) → dedup + first-run-of-day detection │ + ├── Morning (first run) → summary message to #front-sla-alerts ├── Tier 1 → Slack DM assignee or #front-sla-alerts └── Tier 2 → Slack DM Adam ``` @@ -34,9 +40,9 @@ Lambda (Python 3.12, arm64) ## AWS Resources - **Stack:** `front-sla-monitor` (SAM, us-east-1) -- **Lambda:** `front-sla-monitor` — Python 3.12, arm64, 128 MB, 120s timeout, 60-day log retention -- **DynamoDB:** `front-sla-alerts` — tracks alert history per conversation, 7-day TTL -- **EventBridge:** `cron(0/15 * ? * MON-FRI *)` — every 15 min on weekdays +- **Lambda:** `front-sla-monitor` — Python 3.12, arm64, 128 MB, 300s timeout, 60-day log retention +- **DynamoDB:** `front-sla-alerts` — tracks alert history per conversation + monitor state, 7-day TTL +- **EventBridge:** `cron(0/15 12-22 ? * MON-FRI *)` — every 15 min during business hours (UTC range covers EDT/EST) ## Setup @@ -104,3 +110,5 @@ aws logs tail /aws/lambda/front-sla-monitor --follow --region us-east-1 | `ACTION_SLA_MINUTES` | 1440 | Business minutes before Tier 2 alert | | `ADAM_EMAIL` | adam@seahavenind.com | Tier 2 escalation recipient | | `SLACK_ALERT_CHANNEL` | — | Channel ID for broadcast alerts | +| `MONITOR_INBOXES` | Triage,California,West Coast,Central,East Coast,Vendors | Comma-separated inbox names to monitor (empty = all shared) | +| `START_DATE` | 2026-05-14 | Date when monitoring begins (YYYY-MM-DD, Eastern time) | diff --git a/src/monitor/app.py b/src/monitor/app.py index ad044dd..d93d5f7 100644 --- a/src/monitor/app.py +++ b/src/monitor/app.py @@ -15,6 +15,8 @@ EASTERN = ZoneInfo("America/New_York") FRONT_BASE = "https://api2.frontapp.com" SLACK_BASE = "https://slack.com/api" RATE_LIMIT_DELAY = 0.6 +BH_START = 8 +BH_END = 17 _sm = boto3.client("secretsmanager") _ddb = boto3.resource("dynamodb") @@ -210,6 +212,19 @@ def _record_alert(conv_id, tier): ) +def _is_first_run_today(today_str): + resp = _get_table().get_item(Key={"conversationId": "_monitor_state"}) + return resp.get("Item", {}).get("lastRunDate", "") != today_str + + +def _record_run(today_str): + _get_table().update_item( + Key={"conversationId": "_monitor_state"}, + UpdateExpression="SET lastRunDate = :d", + ExpressionAttributeValues={":d": today_str}, + ) + + # --------------------------------------------------------------------------- # Slack message blocks # --------------------------------------------------------------------------- @@ -253,6 +268,34 @@ def _tier2_blocks(conv): ] +def _summary_blocks(tier1_breaches, tier2_breaches): + lines = [] + + if tier2_breaches: + lines.append("*:rotating_light: Over 1 Business Day Without Reply*") + for conv in tier2_breaches: + subject = conv.get("subject", "No subject") + link = f"https://app.frontapp.com/open/{conv.get('id', '')}" + assignee = conv.get("assignee") + who = assignee["email"] if assignee else "Unassigned" + lines.append(f"• <{link}|{subject}> — {who}") + lines.append("") + + if tier1_breaches: + lines.append("*:warning: Over 1 Business Hour Without Reply*") + for conv, email in tier1_breaches: + subject = conv.get("subject", "No subject") + link = f"https://app.frontapp.com/open/{conv.get('id', '')}" + who = email or "Unassigned" + lines.append(f"• <{link}|{subject}> — {who}") + + total = len(tier1_breaches) + len(tier2_breaches) + return [ + {"type": "header", "text": {"type": "plain_text", "text": f":sunrise: Morning SLA Summary — {total} breach{'es' if total != 1 else ''}"}}, + {"type": "section", "text": {"type": "mrkdwn", "text": "\n".join(lines)}}, + ] + + # --------------------------------------------------------------------------- # Handler # --------------------------------------------------------------------------- @@ -262,8 +305,24 @@ def handler(event, context): action_threshold = int(os.environ["ACTION_SLA_MINUTES"]) alert_channel = os.environ["SLACK_ALERT_CHANNEL"] adam_email = os.environ["ADAM_EMAIL"] + start_date = os.environ.get("START_DATE", "") now = datetime.now(timezone.utc) + now_et = now.astimezone(EASTERN) + + if start_date: + start = datetime.strptime(start_date, "%Y-%m-%d").date() + if now_et.date() < start: + logger.info("Before start date %s, skipping", start_date) + return {"skipped": True, "reason": "before_start_date"} + + if now_et.hour < BH_START or now_et.hour >= BH_END: + logger.info("Outside business hours (%s ET), skipping", now_et.strftime("%H:%M")) + return {"skipped": True, "reason": "outside_business_hours"} + + today_str = now_et.strftime("%Y-%m-%d") + first_run = _is_first_run_today(today_str) + _record_run(today_str) monitor_names = os.environ.get("MONITOR_INBOXES", "").strip() if monitor_names: @@ -280,8 +339,8 @@ def handler(event, context): logger.info("Monitoring %d inboxes: %s", len(shared), ", ".join(i.get("name", "?") for i in shared)) - tier1_count = 0 - tier2_count = 0 + tier1_breaches = [] + tier2_breaches = [] seen = set() for inbox in shared: @@ -314,30 +373,49 @@ def handler(event, context): assignee = conv.get("assignee") if elapsed >= action_threshold and not _already_alerted(conv_id, 2): - adam_uid = _resolve_slack_user(adam_email) - target = adam_uid or alert_channel - _send_slack(target, _tier2_blocks(conv), - f"SLA Breach: {conv.get('subject', '')} - 1 day without reply") - _record_alert(conv_id, 2) - tier2_count += 1 - logger.info("Tier 2 alert: %s", conv_id) + tier2_breaches.append(conv) + logger.info("Tier 2 breach: %s", conv_id) elif elapsed >= ack_threshold and not _already_alerted(conv_id, 1): - if assignee and assignee.get("email"): - uid = _resolve_slack_user(assignee["email"]) - target = uid or alert_channel - _send_slack(target, _tier1_blocks(conv, assignee["email"]), - f"SLA Breach: {conv.get('subject', '')} - 1 hour without reply") - else: - _send_slack(alert_channel, _tier1_blocks(conv), - f"SLA Breach: {conv.get('subject', '')} - unassigned, 1 hour without reply") - _record_alert(conv_id, 1) - tier1_count += 1 - logger.info("Tier 1 alert: %s", conv_id) + email = assignee["email"] if assignee and assignee.get("email") else None + tier1_breaches.append((conv, email)) + logger.info("Tier 1 breach: %s", conv_id) except Exception: logger.exception("Error processing conversation %s", conv.get("id", "?")) - result = {"tier1_alerts": tier1_count, "tier2_alerts": tier2_count} + total = len(tier1_breaches) + len(tier2_breaches) + + if first_run and total > 1: + logger.info("Morning summary: %d breaches accumulated overnight", total) + blocks = _summary_blocks(tier1_breaches, tier2_breaches) + _send_slack(alert_channel, blocks, f"Morning SLA Summary: {total} breaches") + if tier2_breaches: + adam_uid = _resolve_slack_user(adam_email) + if adam_uid: + _send_slack(adam_uid, blocks, f"Morning SLA Summary: {total} breaches") + else: + for conv in tier2_breaches: + adam_uid = _resolve_slack_user(adam_email) + target = adam_uid or alert_channel + _send_slack(target, _tier2_blocks(conv), + f"SLA Breach: {conv.get('subject', '')} - 1 day without reply") + + for conv, email in tier1_breaches: + if email: + uid = _resolve_slack_user(email) + target = uid or alert_channel + _send_slack(target, _tier1_blocks(conv, email), + f"SLA Breach: {conv.get('subject', '')} - 1 hour without reply") + else: + _send_slack(alert_channel, _tier1_blocks(conv), + f"SLA Breach: {conv.get('subject', '')} - unassigned, 1 hour without reply") + + for conv in tier2_breaches: + _record_alert(conv["id"], 2) + for conv, _ in tier1_breaches: + _record_alert(conv["id"], 1) + + result = {"tier1_alerts": len(tier1_breaches), "tier2_alerts": len(tier2_breaches)} logger.info("Run complete: %s", result) return result diff --git a/template.yaml b/template.yaml index 1450c77..82d0900 100644 --- a/template.yaml +++ b/template.yaml @@ -28,6 +28,10 @@ Parameters: Type: String Default: "Triage,California,West Coast,Central,East Coast,Vendors" Description: Comma-separated inbox names to monitor (empty = all shared) + StartDate: + Type: String + Default: "2026-05-14" + Description: Date when monitoring begins (YYYY-MM-DD, Eastern time) Globals: Function: @@ -76,6 +80,7 @@ Resources: ACK_SLA_MINUTES: !Ref AckSlaMinutes ACTION_SLA_MINUTES: !Ref ActionSlaMinutes MONITOR_INBOXES: !Ref MonitorInboxes + START_DATE: !Ref StartDate Policies: - DynamoDBCrudPolicy: TableName: !Ref AlertsTable @@ -91,8 +96,8 @@ Resources: SlaCheck: Type: Schedule Properties: - Schedule: cron(0/15 * ? * MON-FRI *) - Description: Check Front conversations for SLA breaches every 15 min on weekdays + Schedule: cron(0/15 12-22 ? * MON-FRI *) + Description: Check Front conversations for SLA breaches every 15 min during business hours Enabled: true Outputs: