From 4870784fbda53d127b9e08248b916c9a82c8cc2e Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 18:05:20 -0400 Subject: [PATCH 01/12] Add self-hosted Grafana stack: EC2, ALB, dashboards-as-code (Phase 5) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The one non-serverless piece — Grafana OSS on a t4g.small (AL2023, ARM64) in the imported seahaven-vpc, fronted by an internet-facing ALB locked by SG to the office CIDRs (no Client VPN exists, so "VPN-only" = office-IP restriction, the syslog-server pattern). Instance in private subnets, reachable only from the ALB SG, administered via SSM Session Manager (no SSH/key pair). grafana_stack.py: ALB (HTTPS, *.seahaven.com cert, open=False so the SG office rules aren't undone by an auto 0.0.0.0/0), instance role (Athena query + Glue read + S3 analytics/athena-results, no static keys), Route53 grafana.seahaven.com alias, gp3 root volume RETAINed, daily DLM snapshot of the tagged instance, and a BucketDeployment that uploads grafana/ to the S3 config prefix. grafana_userdata.sh: install Grafana OSS, pin the Athena datasource plugin, write grafana.ini (root_url grafana.seahaven.com, kiosk embedding), sync provisioning + dashboards from S3 on boot, and a systemd timer re-syncs every 15 min so repo edits land without an instance rebuild. Dashboard (grafana-author agent, grafana/dashboards/apm-work-orders.json, uid apm-wo so the Slack 📊 button resolves): 7 panels — category distribution, escalation summary, action/routine, escalations-by-site, trend time-series over dt (the new capability), filterable WO table (5 template vars, escalation row coloring, CSV export, no APM links), and the mismatch panel. Datasource uid "athena" pinned in the provisioning yaml. Tests: tests/test_grafana_synth.py — ALB admits only the office CIDRs on 443 (caught and fixed a default 0.0.0.0/0 listener rule), instance only-from-ALB, no static keys, scoped instance role + SSM, gp3+retained root volume, daily DLM backup, grafana.seahaven.com alias. 57/57 tests pass; full cdk synth green. --- cdk/assets/grafana_userdata.sh | 92 +++ cdk/cdk.json | 9 +- cdk/stacks/grafana_stack.py | 292 ++++++- grafana/dashboards/apm-work-orders.json | 798 ++++++++++++++++++- grafana/provisioning/datasources/athena.yaml | 1 + tests/test_grafana_synth.py | 177 ++++ 6 files changed, 1353 insertions(+), 16 deletions(-) create mode 100644 cdk/assets/grafana_userdata.sh create mode 100644 tests/test_grafana_synth.py diff --git a/cdk/assets/grafana_userdata.sh b/cdk/assets/grafana_userdata.sh new file mode 100644 index 0000000..bd7c5bd --- /dev/null +++ b/cdk/assets/grafana_userdata.sh @@ -0,0 +1,92 @@ +#!/bin/bash +# Grafana OSS bootstrap for the apm-wo-analysis dashboard host (Amazon Linux 2023, +# ARM64). Idempotent enough to re-run. Config + dashboards are pulled from S3 +# (the repo is the source of truth); a systemd timer re-syncs dashboards so panel +# updates ship by re-uploading to S3 — no instance rebuild. +# +# Templated by CDK: __CONFIG_BUCKET__ / __CONFIG_PREFIX__ / __PLUGIN_VERSION__. +set -euxo pipefail + +CONFIG_BUCKET="__CONFIG_BUCKET__" +CONFIG_PREFIX="__CONFIG_PREFIX__" +PLUGIN_VERSION="__PLUGIN_VERSION__" + +# --- Grafana OSS repo + install --- +cat >/etc/yum.repos.d/grafana.repo <<'REPO' +[grafana] +name=grafana +baseurl=https://rpm.grafana.com +repo_gpgcheck=1 +enabled=1 +gpgcheck=1 +gpgkey=https://rpm.grafana.com/gpg.key +sslverify=1 +REPO +dnf install -y grafana + +# --- Athena datasource plugin (pinned for reproducibility) --- +grafana-cli --pluginsDir /var/lib/grafana/plugins plugins install grafana-athena-datasource "${PLUGIN_VERSION}" + +# --- grafana.ini: behind the ALB at grafana.seahaven.com, kiosk-friendly --- +cat >/etc/grafana/grafana.ini <<'INI' +[server] +protocol = http +http_port = 3000 +root_url = https://grafana.seahaven.com/ +enforce_domain = false + +[security] +# Behind an office-IP-restricted ALB; allow embedding for the kiosk wall display. +allow_embedding = true +cookie_secure = true + +[users] +default_theme = dark + +[analytics] +reporting_enabled = false +check_for_updates = false +INI + +# --- sync provisioning + dashboards from S3 (repo is source of truth) --- +sync_config() { + aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/provisioning/" /etc/grafana/provisioning/ --delete + aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/dashboards/" /var/lib/grafana/dashboards/ --delete + chown -R grafana:grafana /etc/grafana/provisioning /var/lib/grafana/dashboards +} +mkdir -p /var/lib/grafana/dashboards +sync_config + +systemctl daemon-reload +systemctl enable --now grafana-server + +# --- systemd timer: re-sync dashboards every 15 min so repo edits land without a rebuild --- +cat >/usr/local/bin/grafana-config-sync.sh </etc/systemd/system/grafana-config-sync.service <<'SVC' +[Unit] +Description=Sync apm-wo Grafana config/dashboards from S3 +[Service] +Type=oneshot +ExecStart=/usr/local/bin/grafana-config-sync.sh +SVC + +cat >/etc/systemd/system/grafana-config-sync.timer <<'TIMER' +[Unit] +Description=Periodic apm-wo Grafana config sync +[Timer] +OnBootSec=5min +OnUnitActiveSec=15min +[Install] +WantedBy=timers.target +TIMER + +systemctl daemon-reload +systemctl enable --now grafana-config-sync.timer diff --git a/cdk/cdk.json b/cdk/cdk.json index 164a462..5d9f993 100644 --- a/cdk/cdk.json +++ b/cdk/cdk.json @@ -5,6 +5,13 @@ "wildcardCertArn": "arn:aws:acm:us-east-1:328440206208:certificate/a66c0994-90d4-410a-a1d9-5595c2a3fae3", "hostedZoneId": "Z06652411XKH89KTZD3XA", "hostedZoneName": "seahaven.com", - "slackInteractionsDomain": "apm-wo.seahaven.com" + "slackInteractionsDomain": "apm-wo.seahaven.com", + "grafanaVpcId": "vpc-0d3d4b67bd0cf8a68", + "grafanaAzs": ["us-east-1a", "us-east-1b"], + "grafanaPublicSubnetIds": ["subnet-0eea820effe1b3ae5", "subnet-0012f5895182c1580"], + "grafanaPrivateSubnetIds": ["subnet-04e38c507e96f1926", "subnet-0a0b4fc6f296dfba5"], + "grafanaDomain": "grafana.seahaven.com", + "officeCidrs": ["47.21.61.4/32", "96.250.164.146/32"], + "athenaPluginVersion": "2.18.2" } } diff --git a/cdk/stacks/grafana_stack.py b/cdk/stacks/grafana_stack.py index 368bbb1..ee1a713 100644 --- a/cdk/stacks/grafana_stack.py +++ b/cdk/stacks/grafana_stack.py @@ -1,22 +1,290 @@ -"""Grafana stack: VPC import, EC2, ALB, SG, Route53, Athena datasource role. +"""Grafana stack: imported VPC, EC2 (Grafana OSS), ALB, SG, Route53, IAM, backup. -Scaffold — resources are added in Phase 5 of docs/BUILD.md. The running box is -the only non-serverless piece here (self-hosted Grafana OSS on a t4g.small, -ARM64, VPN-only) and carries an OS/Grafana patching + config-backup obligation. -Dashboards are provisioned as code from grafana/ — the running instance is never -the source of truth. +The only non-serverless piece in this repo (self-hosted Grafana OSS on a +t4g.small, ARM64, AL2023). Reachability: an internet-facing ALB whose security +group only admits the office CIDRs — there is no Client VPN in the account, so +"VPN-only" is realized as office-IP restriction (the same pattern syslog-server +uses). The instance sits in private subnets, reachable only from the ALB SG and +administered via SSM Session Manager (no SSH, no key pair). + +Dashboards/datasources are provisioned as code: CDK uploads ``grafana/`` to an S3 +config prefix; instance user-data syncs it on boot and a systemd timer re-syncs, +so the running box is never the source of truth. The gp3 root volume is RETAINed +and snapshotted daily by DLM; ``grafana.db`` lives there. + +Config (cdk.json context): grafanaVpcId, grafanaAzs, grafana{Public,Private}SubnetIds, +grafanaDomain, officeCidrs, wildcardCertArn, hostedZoneId/Name, athenaPluginVersion. """ -from aws_cdk import Stack +import os + +from aws_cdk import ( + CfnTag, + Stack, + Tags, +) +from aws_cdk import ( + aws_certificatemanager as acm, +) +from aws_cdk import ( + aws_dlm as dlm, +) +from aws_cdk import ( + aws_ec2 as ec2, +) +from aws_cdk import ( + aws_elasticloadbalancingv2 as elbv2, +) +from aws_cdk import ( + aws_elasticloadbalancingv2_targets as elbv2_targets, +) +from aws_cdk import ( + aws_iam as iam, +) +from aws_cdk import ( + aws_route53 as route53, +) +from aws_cdk import ( + aws_route53_targets as route53_targets, +) +from aws_cdk import ( + aws_s3 as s3, +) +from aws_cdk import ( + aws_s3_deployment as s3deploy, +) from constructs import Construct +EXPORTS_BUCKET = "apm-wo-analysis-exports-328440206208" +CONFIG_PREFIX = "grafana-config" +BACKUP_TAG = "apm-grafana-backup" +GRAFANA_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "grafana") +USERDATA_PATH = os.path.join( + os.path.dirname(__file__), "..", "assets", "grafana_userdata.sh" +) + class GrafanaStack(Stack): def __init__(self, scope: Construct, construct_id: str, **kwargs) -> None: super().__init__(scope, construct_id, **kwargs) - # Phase 5 — EC2 (Amazon Linux 2023, Grafana OSS via user-data), - # internal ALB (HTTPS, *.seahaven.com ACM cert), - # SG ingress from VPN/office CIDRs only, - # Route53 alias grafana.seahaven.com, - # instance role: Athena + Glue + S3 read (no static keys). TODO + ctx = self.node.try_get_context + domain = ctx("grafanaDomain") + + # Import the shared seahaven-vpc by explicit attributes (no context + # lookup, so the offline synth test needs no AWS credentials). + vpc = ec2.Vpc.from_vpc_attributes( + self, + "SeahavenVpc", + vpc_id=ctx("grafanaVpcId"), + availability_zones=ctx("grafanaAzs"), + public_subnet_ids=ctx("grafanaPublicSubnetIds"), + private_subnet_ids=ctx("grafanaPrivateSubnetIds"), + ) + + # ----- Security groups ----- + alb_sg = ec2.SecurityGroup( + self, + "AlbSg", + vpc=vpc, + description="apm-wo grafana ALB", + allow_all_outbound=True, + ) + for cidr in ctx("officeCidrs"): + alb_sg.add_ingress_rule( + ec2.Peer.ipv4(cidr), ec2.Port.tcp(443), f"HTTPS from office {cidr}" + ) + + instance_sg = ec2.SecurityGroup( + self, + "InstanceSg", + vpc=vpc, + description="apm-wo grafana instance", + allow_all_outbound=True, + ) + instance_sg.add_ingress_rule( + alb_sg, ec2.Port.tcp(3000), "Grafana HTTP from the ALB only" + ) + + # ----- Instance role: Athena query + Glue read + S3 (no static keys) ----- + role = iam.Role( + self, + "GrafanaInstanceRole", + assumed_by=iam.ServicePrincipal("ec2.amazonaws.com"), + managed_policies=[ + # Session Manager admin access; no SSH / bastion / key pair. + iam.ManagedPolicy.from_aws_managed_policy_name( + "AmazonSSMManagedInstanceCore" + ) + ], + ) + role.add_to_policy( + iam.PolicyStatement( + sid="AthenaQuery", + actions=[ + "athena:StartQueryExecution", + "athena:StopQueryExecution", + "athena:GetQueryExecution", + "athena:GetQueryResults", + "athena:GetWorkGroup", + "athena:ListWorkGroups", + ], + resources=[ + f"arn:aws:athena:{self.region}:{self.account}:workgroup/apm-wo-analysis" + ], + ) + ) + role.add_to_policy( + iam.PolicyStatement( + sid="GlueReadOnly", + actions=[ + "glue:GetDatabase", + "glue:GetDatabases", + "glue:GetTable", + "glue:GetTables", + "glue:GetPartition", + "glue:GetPartitions", + ], + resources=[ + f"arn:aws:glue:{self.region}:{self.account}:catalog", + f"arn:aws:glue:{self.region}:{self.account}:database/apm_wo_analysis", + f"arn:aws:glue:{self.region}:{self.account}:table/apm_wo_analysis/*", + ], + ) + ) + bucket = s3.Bucket.from_bucket_name(self, "Exports", EXPORTS_BUCKET) + # Read the analytics snapshots; read+write athena-results (query output). + bucket.grant_read(role, "analytics/*") + bucket.grant_read_write(role, "athena-results/*") + # Config sync reads the grafana-config prefix. + bucket.grant_read(role, f"{CONFIG_PREFIX}/*") + + # ----- Instance (Grafana OSS via user-data) ----- + with open(USERDATA_PATH, encoding="utf-8") as fh: + userdata_script = fh.read() + userdata_script = ( + userdata_script.replace("__CONFIG_BUCKET__", EXPORTS_BUCKET) + .replace("__CONFIG_PREFIX__", CONFIG_PREFIX) + .replace("__PLUGIN_VERSION__", ctx("athenaPluginVersion")) + ) + user_data = ec2.UserData.for_linux() + user_data.add_commands(userdata_script) + + instance = ec2.Instance( + self, + "Grafana", + vpc=vpc, + vpc_subnets=ec2.SubnetSelection( + subnet_type=ec2.SubnetType.PRIVATE_WITH_EGRESS + ), + instance_type=ec2.InstanceType("t4g.small"), + machine_image=ec2.MachineImage.latest_amazon_linux2023( + cpu_type=ec2.AmazonLinuxCpuType.ARM_64 + ), + security_group=instance_sg, + role=role, + user_data=user_data, + require_imdsv2=True, + block_devices=[ + ec2.BlockDevice( + device_name="/dev/xvda", + # RETAIN the gp3 root volume (grafana.db lives here). + volume=ec2.BlockDeviceVolume.ebs( + 20, + volume_type=ec2.EbsDeviceVolumeType.GP3, + delete_on_termination=False, + ), + ) + ], + ) + Tags.of(instance).add(BACKUP_TAG, "true") + + # ----- ALB (internet-facing, office-IP-restricted, HTTPS) ----- + cert = acm.Certificate.from_certificate_arn( + self, "WildcardCert", ctx("wildcardCertArn") + ) + alb = elbv2.ApplicationLoadBalancer( + self, + "Alb", + vpc=vpc, + internet_facing=True, + security_group=alb_sg, + vpc_subnets=ec2.SubnetSelection(subnet_type=ec2.SubnetType.PUBLIC), + ) + listener = alb.add_listener( + "Https", + port=443, + protocol=elbv2.ApplicationProtocol.HTTPS, + certificates=[cert], + # Do NOT auto-open 0.0.0.0/0 on 443 — the ALB SG already scopes + # ingress to the office CIDRs. open=True would undo that. + open=False, + ) + listener.add_targets( + "GrafanaTarget", + port=3000, + protocol=elbv2.ApplicationProtocol.HTTP, + targets=[elbv2_targets.InstanceTarget(instance, 3000)], + health_check=elbv2.HealthCheck( + path="/api/health", healthy_http_codes="200" + ), + ) + + # ----- Route53 alias grafana.seahaven.com -> ALB ----- + zone = route53.HostedZone.from_hosted_zone_attributes( + self, + "SeahavenZone", + hosted_zone_id=ctx("hostedZoneId"), + zone_name=ctx("hostedZoneName"), + ) + route53.ARecord( + self, + "GrafanaAlias", + zone=zone, + record_name=domain.split(".")[0], + target=route53.RecordTarget.from_alias( + route53_targets.LoadBalancerTarget(alb) + ), + ) + + # ----- Dashboards-as-code: upload grafana/ to the S3 config prefix ----- + s3deploy.BucketDeployment( + self, + "GrafanaConfig", + sources=[s3deploy.Source.asset(GRAFANA_DIR)], + destination_bucket=bucket, + destination_key_prefix=CONFIG_PREFIX, + prune=True, + ) + + # ----- Daily DLM snapshot of the (tagged) instance's volume ----- + dlm_role = iam.Role( + self, + "DlmRole", + assumed_by=iam.ServicePrincipal("dlm.amazonaws.com"), + managed_policies=[ + iam.ManagedPolicy.from_aws_managed_policy_name( + "service-role/AWSDataLifecycleManagerServiceRole" + ) + ], + ) + dlm.CfnLifecyclePolicy( + self, + "GrafanaBackup", + description="Daily snapshot of the apm-wo grafana volume", + state="ENABLED", + execution_role_arn=dlm_role.role_arn, + policy_details=dlm.CfnLifecyclePolicy.PolicyDetailsProperty( + resource_types=["INSTANCE"], + target_tags=[CfnTag(key=BACKUP_TAG, value="true")], + schedules=[ + dlm.CfnLifecyclePolicy.ScheduleProperty( + name="daily", + create_rule=dlm.CfnLifecyclePolicy.CreateRuleProperty( + interval=24, interval_unit="HOURS", times=["07:00"] + ), + retain_rule=dlm.CfnLifecyclePolicy.RetainRuleProperty(count=7), + ) + ], + ), + ) diff --git a/grafana/dashboards/apm-work-orders.json b/grafana/dashboards/apm-work-orders.json index 5fdc3c8..ffecd14 100644 --- a/grafana/dashboards/apm-work-orders.json +++ b/grafana/dashboards/apm-work-orders.json @@ -1,7 +1,7 @@ { "uid": "apm-wo", "title": "APM Work Orders", - "description": "Daily APM work-order analysis — breakdown, escalations, trend, filterable WO table, mismatches. Built in Phase 5 (docs/BUILD.md) by the grafana-author agent.", + "description": "Daily APM work-order analysis — breakdown, escalations, trend, filterable WO table, mismatches. Built in Phase 5 by the grafana-author agent. Athena datasource uid='athena'; data grain is one row per WO per daily snapshot partitioned by dt.", "tags": ["apm", "work-orders"], "timezone": "browser", "schemaVersion": 39, @@ -12,7 +12,799 @@ "to": "now" }, "templating": { - "list": [] + "list": [ + { + "name": "dt", + "label": "Snapshot Date", + "description": "Single daily snapshot partition. Defaults to the latest available dt. Most panels filter WHERE dt = '$dt'.", + "type": "query", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "query": { + "rawSql": "SELECT DISTINCT dt FROM apm_wo_analysis.apm_wo_snapshots ORDER BY dt DESC", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + }, + "refresh": 2, + "sort": 0, + "multi": false, + "includeAll": false, + "current": {}, + "options": [], + "hide": 0 + }, + { + "name": "site", + "label": "Site", + "description": "Multi-select. WO table applies: AND ('${site:raw}' = 'All' OR site IN (${site:singlequote}))", + "type": "query", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "query": { + "rawSql": "SELECT DISTINCT site FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND site<>'' ORDER BY 1", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + }, + "refresh": 2, + "sort": 1, + "multi": true, + "includeAll": true, + "allValue": "All", + "current": {}, + "options": [], + "hide": 0 + }, + { + "name": "department", + "label": "Department", + "description": "Multi-select. WO table applies: AND ('${department:raw}' = 'All' OR department IN (${department:singlequote}))", + "type": "query", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "query": { + "rawSql": "SELECT DISTINCT department FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND department<>'' ORDER BY 1", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + }, + "refresh": 2, + "sort": 1, + "multi": true, + "includeAll": true, + "allValue": "All", + "current": {}, + "options": [], + "hide": 0 + }, + { + "name": "category", + "label": "Category", + "description": "Multi-select. WO table applies: AND ('${category:raw}' = 'All' OR category IN (${category:singlequote}))", + "type": "query", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "query": { + "rawSql": "SELECT DISTINCT category FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND category<>'' ORDER BY 1", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + }, + "refresh": 2, + "sort": 1, + "multi": true, + "includeAll": true, + "allValue": "All", + "current": {}, + "options": [], + "hide": 0 + }, + { + "name": "wo_status", + "label": "WO Status", + "description": "Multi-select. WO table applies: AND ('${wo_status:raw}' = 'All' OR wo_status IN (${wo_status:singlequote}))", + "type": "query", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "query": { + "rawSql": "SELECT DISTINCT wo_status FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND wo_status<>'' ORDER BY 1", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + }, + "refresh": 2, + "sort": 1, + "multi": true, + "includeAll": true, + "allValue": "All", + "current": {}, + "options": [], + "hide": 0 + }, + { + "name": "hold_reason", + "label": "Hold Reason", + "description": "Multi-select. WO table applies: AND ('${hold_reason:raw}' = 'All' OR hold_reason IN (${hold_reason:singlequote}))", + "type": "query", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "query": { + "rawSql": "SELECT DISTINCT hold_reason FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND hold_reason<>'' ORDER BY 1", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + }, + "refresh": 2, + "sort": 1, + "multi": true, + "includeAll": true, + "allValue": "All", + "current": {}, + "options": [], + "hide": 0 + } + ] }, - "panels": [] + "panels": [ + { + "id": 1, + "type": "bar-chart", + "title": "Category Distribution", + "description": "Count of work orders by classification category for the selected snapshot date. Sorted descending by volume. Filter using the template variables above.", + "gridPos": { "x": 0, "y": 0, "w": 14, "h": 9 }, + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "rawSql": "SELECT category, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1 ORDER BY 2 DESC", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + } + ], + "options": { + "orientation": "horizontal", + "barRadius": 0, + "groupWidth": 0.7, + "showValue": "always", + "stacking": "none", + "tooltip": { "mode": "single", "sort": "none" }, + "legend": { "showLegend": false, "displayMode": "list", "placement": "bottom" } + }, + "fieldConfig": { + "defaults": { + "color": { "mode": "palette-classic" }, + "custom": { + "axisBorderShow": false, + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "fillOpacity": 80, + "gradientMode": "none", + "hideFrom": { "legend": false, "tooltip": false, "viz": false }, + "lineWidth": 1, + "scaleDistribution": { "type": "linear" }, + "thresholdsStyle": { "mode": "off" } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { "color": "green", "value": null }, + { "color": "red", "value": 80 } + ] + } + }, + "overrides": [] + } + }, + { + "id": 2, + "type": "piechart", + "title": "Escalation Summary", + "description": "Distribution across escalation categories (1st, 2nd, 3rd Escalation, SIM Ticket, Other Escalation) for the selected snapshot date. 3rd Escalation is highlighted red.", + "gridPos": { "x": 14, "y": 0, "w": 10, "h": 9 }, + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "rawSql": "SELECT category, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true GROUP BY 1 ORDER BY 2 DESC", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + } + ], + "options": { + "pieType": "pie", + "displayLabels": ["name", "value"], + "tooltip": { "mode": "single", "sort": "none" }, + "legend": { "showLegend": true, "displayMode": "table", "placement": "right", "values": ["value", "percent"] } + }, + "fieldConfig": { + "defaults": { + "color": { "mode": "palette-classic" }, + "custom": { + "hideFrom": { "legend": false, "tooltip": false, "viz": false } + }, + "mappings": [] + }, + "overrides": [ + { + "matcher": { "id": "byName", "options": "3rd Escalation" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "red", "mode": "fixed" } } + ] + }, + { + "matcher": { "id": "byName", "options": "2nd Escalation" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "orange", "mode": "fixed" } } + ] + }, + { + "matcher": { "id": "byName", "options": "1st Escalation" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "yellow", "mode": "fixed" } } + ] + }, + { + "matcher": { "id": "byName", "options": "SIM Ticket" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "purple", "mode": "fixed" } } + ] + } + ] + } + }, + { + "id": 3, + "type": "piechart", + "title": "Action-Needed vs Routine", + "description": "Action-needed WOs require attention (escalations, scheduling, vendor/report waits, status inquiries, vendor no-shows). Routine WOs are on track. Counts are for the selected snapshot date.", + "gridPos": { "x": 0, "y": 9, "w": 8, "h": 8 }, + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "rawSql": "SELECT CASE WHEN is_action=true THEN 'Action Needed' ELSE 'Routine' END AS action_type, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + } + ], + "options": { + "pieType": "donut", + "displayLabels": ["name", "percent"], + "tooltip": { "mode": "single", "sort": "none" }, + "legend": { "showLegend": true, "displayMode": "list", "placement": "bottom", "values": ["value", "percent"] } + }, + "fieldConfig": { + "defaults": { + "color": { "mode": "palette-classic" }, + "custom": { + "hideFrom": { "legend": false, "tooltip": false, "viz": false } + }, + "mappings": [] + }, + "overrides": [ + { + "matcher": { "id": "byName", "options": "Action Needed" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "semi-dark-orange", "mode": "fixed" } } + ] + }, + { + "matcher": { "id": "byName", "options": "Routine" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "green", "mode": "fixed" } } + ] + } + ] + } + }, + { + "id": 4, + "type": "bar-chart", + "title": "Escalations by Site", + "description": "Total escalations per site for the selected snapshot date. Includes all escalation categories.", + "gridPos": { "x": 8, "y": 9, "w": 16, "h": 8 }, + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "rawSql": "SELECT site, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true AND site<>'' GROUP BY 1 ORDER BY 2 DESC", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + } + ], + "options": { + "orientation": "vertical", + "barRadius": 0, + "groupWidth": 0.7, + "showValue": "always", + "stacking": "none", + "tooltip": { "mode": "single", "sort": "none" }, + "legend": { "showLegend": false, "displayMode": "list", "placement": "bottom" }, + "xTickLabelRotation": -45, + "xTickLabelMaxLength": 12 + }, + "fieldConfig": { + "defaults": { + "color": { "mode": "fixed", "fixedColor": "semi-dark-red" }, + "custom": { + "axisBorderShow": false, + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "Escalations", + "axisPlacement": "auto", + "fillOpacity": 80, + "gradientMode": "none", + "hideFrom": { "legend": false, "tooltip": false, "viz": false }, + "lineWidth": 1, + "scaleDistribution": { "type": "linear" }, + "thresholdsStyle": { "mode": "off" } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { "color": "green", "value": null } + ] + } + }, + "overrides": [] + } + }, + { + "id": 5, + "type": "timeseries", + "title": "Trend Over Time — Escalations & Action-Needed per Day", + "description": "Runs across all partitions (no $dt filter) to show daily escalation and action-needed volume over time. Use the dashboard time range picker to zoom in. This is the capability the legacy Sheet never had.", + "gridPos": { "x": 0, "y": 17, "w": 24, "h": 9 }, + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "rawSql": "SELECT date_parse(dt, '%Y-%m-%d') AS time, SUM(CAST(is_escalation AS INTEGER)) AS escalations, SUM(CAST(is_action AS INTEGER)) AS action_needed FROM apm_wo_analysis.apm_wo_snapshots GROUP BY 1 ORDER BY 1", + "format": "timeSeries", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + } + ], + "options": { + "tooltip": { "mode": "multi", "sort": "desc" }, + "legend": { "showLegend": true, "displayMode": "list", "placement": "bottom", "calcs": ["mean", "max", "last"] } + }, + "fieldConfig": { + "defaults": { + "color": { "mode": "palette-classic" }, + "custom": { + "axisBorderShow": false, + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "Work Orders", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 10, + "gradientMode": "none", + "hideFrom": { "legend": false, "tooltip": false, "viz": false }, + "insertNulls": false, + "lineInterpolation": "linear", + "lineWidth": 2, + "pointSize": 5, + "scaleDistribution": { "type": "linear" }, + "showPoints": "auto", + "spanNulls": false, + "stacking": { "group": "A", "mode": "none" }, + "thresholdsStyle": { "mode": "off" } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { "color": "green", "value": null } + ] + }, + "unit": "short" + }, + "overrides": [ + { + "matcher": { "id": "byName", "options": "escalations" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "semi-dark-red", "mode": "fixed" } }, + { "id": "displayName", "value": "Escalations" } + ] + }, + { + "matcher": { "id": "byName", "options": "action_needed" }, + "properties": [ + { "id": "color", "value": { "fixedColor": "semi-dark-orange", "mode": "fixed" } }, + { "id": "displayName", "value": "Action Needed" } + ] + } + ] + } + }, + { + "id": 6, + "type": "table", + "title": "Work Order Detail Table", + "description": "Filterable, exportable table of all work orders for the selected snapshot date. Multi-value template variables are applied via: AND ('${variable:raw}' = 'All' OR column IN (${variable:singlequote})). Escalation rows are color-coded: 3rd Escalation = red, 2nd Escalation = orange, 1st Escalation = yellow. WO numbers are plain text (no APM deep-link per project decision). Use the Download CSV button (table header menu) to export. last_comment is wrapped for readability.", + "gridPos": { "x": 0, "y": 26, "w": 24, "h": 14 }, + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "rawSql": "SELECT wo_number, site, department, category, wo_status, hold_reason, wo_description, last_comment FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND ('${site:raw}' = 'All' OR site IN (${site:singlequote})) AND ('${department:raw}' = 'All' OR department IN (${department:singlequote})) AND ('${category:raw}' = 'All' OR category IN (${category:singlequote})) AND ('${wo_status:raw}' = 'All' OR wo_status IN (${wo_status:singlequote})) AND ('${hold_reason:raw}' = 'All' OR hold_reason IN (${hold_reason:singlequote})) ORDER BY category, site", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + } + ], + "options": { + "frameIndex": 0, + "showHeader": true, + "sortBy": [], + "footer": { + "show": false, + "reducer": ["sum"], + "fields": "", + "enablePagination": false + } + }, + "fieldConfig": { + "defaults": { + "color": { "mode": "thresholds" }, + "custom": { + "align": "left", + "cellOptions": { "type": "auto" }, + "inspect": false, + "filterable": true, + "minWidth": 80, + "width": 0 + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { "color": "text", "value": null } + ] + } + }, + "overrides": [ + { + "matcher": { "id": "byName", "options": "wo_number" }, + "properties": [ + { "id": "displayName", "value": "WO #" }, + { "id": "custom.width", "value": 100 } + ] + }, + { + "matcher": { "id": "byName", "options": "site" }, + "properties": [ + { "id": "displayName", "value": "Site" }, + { "id": "custom.width", "value": 80 } + ] + }, + { + "matcher": { "id": "byName", "options": "department" }, + "properties": [ + { "id": "displayName", "value": "Dept" }, + { "id": "custom.width", "value": 80 } + ] + }, + { + "matcher": { "id": "byName", "options": "category" }, + "properties": [ + { "id": "displayName", "value": "Category" }, + { "id": "custom.width", "value": 160 }, + { + "id": "mappings", + "value": [ + { + "type": "value", + "options": { + "3rd Escalation": { + "color": "dark-red", + "index": 0 + }, + "2nd Escalation": { + "color": "dark-orange", + "index": 1 + }, + "1st Escalation": { + "color": "dark-yellow", + "index": 2 + }, + "SIM Ticket": { + "color": "dark-purple", + "index": 3 + }, + "Other Escalation": { + "color": "orange", + "index": 4 + } + } + } + ] + }, + { + "id": "custom.cellOptions", + "value": { "type": "color-text" } + } + ] + }, + { + "matcher": { "id": "byName", "options": "wo_status" }, + "properties": [ + { "id": "displayName", "value": "Status" }, + { "id": "custom.width", "value": 80 } + ] + }, + { + "matcher": { "id": "byName", "options": "hold_reason" }, + "properties": [ + { "id": "displayName", "value": "Hold Reason" }, + { "id": "custom.width", "value": 120 } + ] + }, + { + "matcher": { "id": "byName", "options": "wo_description" }, + "properties": [ + { "id": "displayName", "value": "Description" }, + { "id": "custom.width", "value": 220 } + ] + }, + { + "matcher": { "id": "byName", "options": "last_comment" }, + "properties": [ + { "id": "displayName", "value": "Last Comment" }, + { + "id": "custom.cellOptions", + "value": { "type": "auto", "wrapText": true } + }, + { "id": "custom.width", "value": 420 }, + { "id": "custom.minWidth", "value": 200 } + ] + } + ] + } + }, + { + "id": 7, + "type": "table", + "title": "Mismatch Panel — Comment vs Structured-State Contradictions", + "description": "Work orders where the classified comment intent contradicts the structured WO state (e.g. comment says 'completed' but WO is on a REPORT/VENDOR/SCHEDULING hold, or comment says 'scheduled' while status is IP). Non-empty mismatch column only. Use these rows for manual review before the next export.", + "gridPos": { "x": 0, "y": 40, "w": 24, "h": 8 }, + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "grafana-athena-datasource", + "uid": "athena" + }, + "rawSql": "SELECT wo_number, site, category, wo_status, hold_reason, mismatch FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND mismatch<>'' ORDER BY category, site", + "format": "table", + "connectionArgs": { + "catalog": "AwsDataCatalog", + "database": "apm_wo_analysis", + "region": "us-east-1" + } + } + ], + "options": { + "frameIndex": 0, + "showHeader": true, + "sortBy": [], + "footer": { + "show": false, + "reducer": ["sum"], + "fields": "", + "enablePagination": false + } + }, + "fieldConfig": { + "defaults": { + "color": { "mode": "thresholds" }, + "custom": { + "align": "left", + "cellOptions": { "type": "auto" }, + "inspect": false, + "filterable": true, + "minWidth": 80, + "width": 0 + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { "color": "text", "value": null } + ] + } + }, + "overrides": [ + { + "matcher": { "id": "byName", "options": "wo_number" }, + "properties": [ + { "id": "displayName", "value": "WO #" }, + { "id": "custom.width", "value": 100 } + ] + }, + { + "matcher": { "id": "byName", "options": "site" }, + "properties": [ + { "id": "displayName", "value": "Site" }, + { "id": "custom.width", "value": 80 } + ] + }, + { + "matcher": { "id": "byName", "options": "category" }, + "properties": [ + { "id": "displayName", "value": "Category" }, + { "id": "custom.width", "value": 160 }, + { + "id": "mappings", + "value": [ + { + "type": "value", + "options": { + "3rd Escalation": { + "color": "dark-red", + "index": 0 + }, + "2nd Escalation": { + "color": "dark-orange", + "index": 1 + }, + "1st Escalation": { + "color": "dark-yellow", + "index": 2 + }, + "SIM Ticket": { + "color": "dark-purple", + "index": 3 + }, + "Other Escalation": { + "color": "orange", + "index": 4 + } + } + } + ] + }, + { + "id": "custom.cellOptions", + "value": { "type": "color-text" } + } + ] + }, + { + "matcher": { "id": "byName", "options": "wo_status" }, + "properties": [ + { "id": "displayName", "value": "WO Status" }, + { "id": "custom.width", "value": 90 } + ] + }, + { + "matcher": { "id": "byName", "options": "hold_reason" }, + "properties": [ + { "id": "displayName", "value": "Hold Reason" }, + { "id": "custom.width", "value": 120 } + ] + }, + { + "matcher": { "id": "byName", "options": "mismatch" }, + "properties": [ + { "id": "displayName", "value": "Mismatch Description" }, + { + "id": "custom.cellOptions", + "value": { "type": "auto", "wrapText": true } + }, + { "id": "custom.width", "value": 500 }, + { "id": "custom.minWidth", "value": 200 }, + { "id": "color", "value": { "mode": "fixed", "fixedColor": "semi-dark-orange" } } + ] + } + ] + } + } + ] } diff --git a/grafana/provisioning/datasources/athena.yaml b/grafana/provisioning/datasources/athena.yaml index 5831d8b..58072bf 100644 --- a/grafana/provisioning/datasources/athena.yaml +++ b/grafana/provisioning/datasources/athena.yaml @@ -4,6 +4,7 @@ apiVersion: 1 datasources: - name: Athena type: grafana-athena-datasource + uid: athena isDefault: true jsonData: authType: ec2_iam_role diff --git a/tests/test_grafana_synth.py b/tests/test_grafana_synth.py new file mode 100644 index 0000000..42f29b1 --- /dev/null +++ b/tests/test_grafana_synth.py @@ -0,0 +1,177 @@ +"""Synth-level assertions for the Grafana stack (Phase 5). + +Synthesizes ``apm-wo-analysis-grafana`` and asserts the security posture that +can't be eyeballed: the ALB only admits the office CIDRs on 443 (never +0.0.0.0/0), the instance only takes traffic from the ALB SG, the instance role +carries no static keys and only scoped Athena/Glue-read/S3 access, the root +volume is gp3 + retained, a daily DLM backup exists, and grafana.seahaven.com +aliases the ALB. No AWS, no Docker (bundling skipped). + +Run with the repo venv: + + python -m pytest tests/test_grafana_synth.py -q +""" + +import json +import sys +from pathlib import Path + +import aws_cdk as cdk +from aws_cdk.assertions import Match, Template + +CDK_DIR = Path(__file__).resolve().parents[1] / "cdk" +sys.path.insert(0, str(CDK_DIR)) + +from stacks.grafana_stack import GrafanaStack # noqa: E402 + +OFFICE_CIDRS = {"47.21.61.4/32", "96.250.164.146/32"} + + +def _cdk_context() -> dict: + ctx = json.loads((CDK_DIR / "cdk.json").read_text())["context"] + ctx["aws:cdk:bundling-stacks"] = [] + return ctx + + +def _template() -> Template: + app = cdk.App(context=_cdk_context()) + stack = GrafanaStack( + app, + "apm-wo-analysis-grafana", + env=cdk.Environment(account="328440206208", region="us-east-1"), + ) + return Template.from_stack(stack) + + +def _all_cidr_ingress(t: Template): + """Every CIDR-based ingress rule, inline on SGs and standalone, as + (cidr, from_port, to_port) tuples.""" + rules = [] + for sg in t.find_resources("AWS::EC2::SecurityGroup").values(): + for r in sg["Properties"].get("SecurityGroupIngress", []): + if "CidrIp" in r: + rules.append((r["CidrIp"], r.get("FromPort"), r.get("ToPort"))) + for ing in t.find_resources("AWS::EC2::SecurityGroupIngress").values(): + p = ing["Properties"] + if "CidrIp" in p: + rules.append((p["CidrIp"], p.get("FromPort"), p.get("ToPort"))) + return rules + + +def test_alb_only_admits_office_cidrs_on_443(): + rules = _all_cidr_ingress(_template()) + cidrs_443 = {c for c, fp, tp in rules if fp == 443 and tp == 443} + assert cidrs_443 == OFFICE_CIDRS, ( + f"443 ingress should be office-only, got {cidrs_443}" + ) + # Nothing anywhere may be open to the world. + assert all(c != "0.0.0.0/0" for c, _, _ in rules), "found a 0.0.0.0/0 ingress" + + +def test_instance_only_reachable_from_alb_on_3000(): + # The instance SG ingress on 3000 is a SourceSecurityGroup rule, not a CIDR. + _template().has_resource_properties( + "AWS::EC2::SecurityGroupIngress", + Match.object_like( + { + "FromPort": 3000, + "ToPort": 3000, + "SourceSecurityGroupId": Match.any_value(), + } + ), + ) + + +def test_alb_internet_facing_https_listener(): + t = _template() + t.has_resource_properties( + "AWS::ElasticLoadBalancingV2::LoadBalancer", {"Scheme": "internet-facing"} + ) + t.has_resource_properties( + "AWS::ElasticLoadBalancingV2::Listener", + Match.object_like( + {"Port": 443, "Protocol": "HTTPS", "Certificates": Match.any_value()} + ), + ) + + +def test_no_static_keys_in_stack(): + t = _template() + t.resource_count_is("AWS::IAM::User", 0) + t.resource_count_is("AWS::IAM::AccessKey", 0) + + +def test_instance_role_scoped_and_uses_ssm(): + t = _template() + # Session Manager (no SSH) — the SSM managed policy is attached. + t.has_resource_properties( + "AWS::IAM::Role", + Match.object_like( + { + "ManagedPolicyArns": Match.array_with( + [ + { + "Fn::Join": [ + "", + Match.array_with( + [":iam::aws:policy/AmazonSSMManagedInstanceCore"] + ), + ] + } + ] + ) + } + ), + ) + # The instance role must not be able to write the catalog or run wide Athena. + for policy in t.find_resources("AWS::IAM::Policy").values(): + for stmt in policy["Properties"]["PolicyDocument"]["Statement"]: + actions = stmt.get("Action", []) + actions = [actions] if isinstance(actions, str) else actions + for a in actions: + if isinstance(a, str): + assert a not in ("glue:*", "athena:*", "s3:*", "*"), ( + f"too broad: {a}" + ) + assert not a.startswith("glue:Create"), f"no Glue writes: {a}" + assert not a.startswith("glue:Update"), f"no Glue writes: {a}" + + +def test_root_volume_gp3_and_retained(): + _template().has_resource_properties( + "AWS::EC2::Instance", + Match.object_like( + { + "BlockDeviceMappings": Match.array_with( + [ + Match.object_like( + { + "Ebs": Match.object_like( + {"VolumeType": "gp3", "DeleteOnTermination": False} + ) + } + ) + ] + ) + } + ), + ) + + +def test_daily_dlm_backup_enabled(): + _template().has_resource_properties( + "AWS::DLM::LifecyclePolicy", + Match.object_like( + { + "State": "ENABLED", + "PolicyDetails": Match.object_like({"ResourceTypes": ["INSTANCE"]}), + } + ), + ) + + +def test_route53_alias_for_grafana(): + _template().has_resource_properties( + "AWS::Route53::RecordSet", + Match.object_like({"Type": "A", "Name": "grafana.seahaven.com."}), + ) From c04c239774d4c4820c9fc5a8f32fa3b32ea3da1b Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 18:06:10 -0400 Subject: [PATCH 02/12] Update README status for Phase 5 Grafana stack --- README.md | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 327f1b4..b5ff1f8 100644 --- a/README.md +++ b/README.md @@ -142,7 +142,7 @@ cdk deploy apm-wo-analysis-grafana # EC2, ALB, SG, Route53, datasource role ## Status -**Phase 4 — Slack surfaces (in review).** Build-out proceeds per +**Phase 5 — Grafana (in review); build complete, docs remain.** Build-out per [`docs/BUILD.md`](./docs/BUILD.md): ingestion → classifier → Glue/Athena → Slack → Grafana → docs. @@ -159,4 +159,10 @@ cdk deploy apm-wo-analysis-grafana # EC2, ALB, SG, Route53, datasource role 3rd-escalation alert, drill-down modals on `apm-wo.seahaven.com`) — implemented and `cdk synth`-green; PR #10 open (stacked on Phase 3). App manifest in `slack/manifest.yaml`. -- **Phases 5–6** (Grafana, final docs) — not started. +- **Phase 5** self-hosted Grafana (EC2 + ALB on `grafana.seahaven.com`, + office-IP-restricted, 7-panel dashboard as code) — implemented and + `cdk synth`-green; PR #11 open (stacked on Phase 4). +- **Phase 6** (final docs / Confluence / runbook) — not started. + +All phase PRs are stacked (#6→#7→#8→#9→#10→#11) and **unmerged**; nothing is +deployed yet. Cross-review and `/security-review` are outstanding across the stack. From f193754c27befc55199246396aa66967847a7754 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 18:29:16 -0400 Subject: [PATCH 03/12] Fix deploy-time failures found in prod testing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two issues only a real deploy/run surfaced (synth + offline tests passed): 1. Classifier exceeded Lambda's 250 MB unzipped limit (bundled awswrangler + pandas + pyarrow + numpy). Move them to the AWS-managed SDK-for-pandas layer (AWSSDKPandas-Python312-Arm64:27, awswrangler 3.16.1, pre-stripped to fit); bundle only openpyxl. Drop the unused anthropic SDK — _call_haiku uses stdlib urllib. Function package now ~890 KB. 2. Slack rejected the daily post with invalid_blocks: every category drill button shared action_id "drill_category". Qualify it as "drill_category:" for uniqueness; the interactions handler now matches on the prefix. Add a regression test asserting all daily-summary action_ids are unique. Verified in prod: classifier writes Parquet + summary.json + details.json; slack-post posts the daily summary + 3rd-escalation alert; the interactions endpoint (apm-wo.seahaven.com) returns 401 on a bad signature. 58/58 tests pass. NOTE: these fixes sit on the phase-5 branch but logically belong to earlier phases — the layer fix to #8 (classifier), the Slack fix to #10 — and must be moved/cherry-picked there before those PRs merge independently. See cleanup. --- cdk/stacks/pipeline_stack.py | 14 ++++++++++++++ lambdas/classifier/requirements.txt | 6 ++++-- lambdas/slack_post/blockkit.py | 5 ++++- lambdas/slack_post/interactions.py | 5 ++++- tests/test_blockkit.py | 17 ++++++++++++++++- 5 files changed, 42 insertions(+), 5 deletions(-) diff --git a/cdk/stacks/pipeline_stack.py b/cdk/stacks/pipeline_stack.py index 8b3a36e..aa38888 100644 --- a/cdk/stacks/pipeline_stack.py +++ b/cdk/stacks/pipeline_stack.py @@ -64,6 +64,11 @@ from constructs import Construct GLUE_DATABASE = "apm_wo_analysis" GLUE_TABLE = "apm_wo_snapshots" ATHENA_WORKGROUP = "apm-wo-analysis" +# AWS-managed SDK-for-pandas layer (awswrangler 3.16.1, py3.12, arm64). Provides +# awswrangler/pandas/pyarrow/numpy pre-stripped to fit the Lambda size limit. +AWSSDKPANDAS_LAYER_ARN = ( + "arn:aws:lambda:us-east-1:336392948345:layer:AWSSDKPandas-Python312-Arm64:27" +) ANTHROPIC_SECRET = "apm-wo-analysis/anthropic-api-key" SLACK_SECRET = "apm-wo-analysis/slack-credentials" GRAFANA_URL_PARAM = "/apm-wo-analysis/grafana-base-url" @@ -231,6 +236,15 @@ class PipelineStack(Stack): timeout=Duration.seconds(120), log_group=classifier_logs, environment={"APM_HAIKU_FALLBACK": "on"}, + # awswrangler/pandas/pyarrow/numpy come from the AWS-managed + # SDK-for-pandas layer (pre-stripped to fit the 250 MB unzipped + # limit, which bundling them ourselves blows). The function package + # only bundles openpyxl; boto3 is in the runtime, urllib is stdlib. + layers=[ + lambda_.LayerVersion.from_layer_version_arn( + self, "PandasLayer", AWSSDKPANDAS_LAYER_ARN + ) + ], code=lambda_.Code.from_asset( os.path.join(LAMBDAS_DIR, "classifier"), bundling=BundlingOptions( diff --git a/lambdas/classifier/requirements.txt b/lambdas/classifier/requirements.txt index 4f3a94d..2ac82aa 100644 --- a/lambdas/classifier/requirements.txt +++ b/lambdas/classifier/requirements.txt @@ -1,3 +1,5 @@ -awswrangler>=3.9.0 +# awswrangler + pandas/pyarrow/numpy come from the AWS-managed SDK-for-pandas +# Lambda layer (see pipeline_stack.py) — bundling them here blows the 250 MB +# unzipped limit. The Haiku fallback uses stdlib urllib, so no anthropic SDK. +# Only openpyxl (xlsx parsing) is bundled into the function package. openpyxl>=3.1.0 -anthropic>=0.40.0 diff --git a/lambdas/slack_post/blockkit.py b/lambdas/slack_post/blockkit.py index 2823976..3cab1bf 100644 --- a/lambdas/slack_post/blockkit.py +++ b/lambdas/slack_post/blockkit.py @@ -282,8 +282,11 @@ def build_daily_summary( for cat, count in drill_cats: short_label = cat.replace(" Escalation", " Esc.").replace("Awaiting ", "") short_label = short_label[:20] # button text kept concise + # action_id must be UNIQUE per message (Slack rejects duplicates), so + # qualify it with the category; the interactions handler matches on the + # "drill_category:" prefix and reads the filter value from `value`. button_elements.append( - _button(f"{short_label} ({count})", "drill_category", cat) + _button(f"{short_label} ({count})", f"drill_category:{cat}", cat) ) # Dashboard link button always present (no action_id — url button). diff --git a/lambdas/slack_post/interactions.py b/lambdas/slack_post/interactions.py index 647f3fe..b664fda 100644 --- a/lambdas/slack_post/interactions.py +++ b/lambdas/slack_post/interactions.py @@ -49,7 +49,10 @@ def handler(event, context): return {"statusCode": 200, "body": ""} action = (payload.get("actions") or [{}])[0] - field = _FILTER_FIELD.get(action.get("action_id")) + # action_id is qualified for Slack uniqueness, e.g. "drill_category:Report / + # Docs Needed" — match on the prefix before ":"; the filter value is in `value`. + action_kind = (action.get("action_id") or "").split(":", 1)[0] + field = _FILTER_FIELD.get(action_kind) value = action.get("value") trigger_id = payload.get("trigger_id") if not field or not value or not trigger_id: diff --git a/tests/test_blockkit.py b/tests/test_blockkit.py index 02eb1d7..6a2fb1d 100644 --- a/tests/test_blockkit.py +++ b/tests/test_blockkit.py @@ -151,6 +151,20 @@ class TestDailySummaryStructure: blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL) assert any(b.get("type") == "header" for b in blocks) + def test_action_ids_are_unique(self): + # Slack rejects a message with duplicate action_ids across its elements + # (regression: all drill buttons once shared action_id "drill_category"). + blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL) + action_ids = [ + el["action_id"] + for b in blocks + for el in b.get("elements", []) + if isinstance(el, dict) and "action_id" in el + ] + assert len(action_ids) == len(set(action_ids)), ( + f"duplicate action_id(s): {action_ids}" + ) + def test_header_contains_date(self): blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL) header = next(b for b in blocks if b.get("type") == "header") @@ -191,7 +205,8 @@ class TestDailySummaryStructure: if block.get("type") != "actions": continue for elem in block.get("elements", []): - if elem.get("action_id") == "drill_category": + # action_id is qualified for uniqueness: "drill_category:". + if str(elem.get("action_id", "")).startswith("drill_category:"): drill_found = True assert "value" in elem assert drill_found, "expected at least one drill_category action button" From ea54cb1e608ba172f3035768c849b464b8c6892c Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 18:43:08 -0400 Subject: [PATCH 04/12] Fix Grafana bootstrap: grafana-cli --homepath + valid plugin version MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found on first boot (cloud-init errored, grafana-server never started): - grafana-cli needs --homepath=/usr/share/grafana or it can't find config defaults; under set -e that aborted the whole bootstrap. - pinned plugin version 2.18.2 doesn't exist (conflated with awswrangler's version) — grafana-athena-datasource latest is 3.2.0. Verified by running the corrected bootstrap on the instance via SSM: plugin installs, grafana-server active, /api/health 200, ALB target healthy. NOTE: the running instance was repaired in-place (the user-data change updated the launch template but did not replace the instance). The committed user-data is now correct, so a fresh launch boots clean — a one-time clean instance replacement should validate that before prod sign-off. --- cdk/assets/grafana_userdata.sh | 4 +++- cdk/cdk.json | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/cdk/assets/grafana_userdata.sh b/cdk/assets/grafana_userdata.sh index bd7c5bd..f67ade6 100644 --- a/cdk/assets/grafana_userdata.sh +++ b/cdk/assets/grafana_userdata.sh @@ -25,7 +25,9 @@ REPO dnf install -y grafana # --- Athena datasource plugin (pinned for reproducibility) --- -grafana-cli --pluginsDir /var/lib/grafana/plugins plugins install grafana-athena-datasource "${PLUGIN_VERSION}" +# --homepath is required or grafana-cli can't find its config defaults. +grafana-cli --homepath=/usr/share/grafana --pluginsDir=/var/lib/grafana/plugins \ + plugins install grafana-athena-datasource "${PLUGIN_VERSION}" # --- grafana.ini: behind the ALB at grafana.seahaven.com, kiosk-friendly --- cat >/etc/grafana/grafana.ini <<'INI' diff --git a/cdk/cdk.json b/cdk/cdk.json index 5d9f993..bf1cadf 100644 --- a/cdk/cdk.json +++ b/cdk/cdk.json @@ -12,6 +12,6 @@ "grafanaPrivateSubnetIds": ["subnet-04e38c507e96f1926", "subnet-0a0b4fc6f296dfba5"], "grafanaDomain": "grafana.seahaven.com", "officeCidrs": ["47.21.61.4/32", "96.250.164.146/32"], - "athenaPluginVersion": "2.18.2" + "athenaPluginVersion": "3.2.0" } } From 68f3acded898f7a5912cca92d253c4212d33fed8 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 18:49:41 -0400 Subject: [PATCH 05/12] Fix dashboard rendering: barchart panel type + metadata off the table prefix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two issues found loading the deployed dashboard: 1. Panels used type "bar-chart" (hyphenated); Grafana's core panel is "barchart" — hence "plugin bar-chart required". Fixed both panels. 2. ALL panels showed "no data" because Athena failed with HIVE_BAD_DATA: the classifier wrote summary.json/details.json INTO analytics/dt=*/ — the same prefix the Glue table scans — so Athena tried to read the JSON as Parquet and every query failed. Move the metadata to a separate meta/dt=*/ prefix: classifier writes there (grant_read_write meta/*), the Slack Lambdas read there (read_meta_json, grant_read meta/*), and analytics/ holds only Parquet. Verified: the category GROUP BY query now succeeds against Athena. --- cdk/stacks/pipeline_stack.py | 6 +++++- grafana/dashboards/apm-work-orders.json | 4 ++-- lambdas/classifier/handler.py | 9 ++++++--- lambdas/slack_post/handler.py | 6 +++--- lambdas/slack_post/interactions.py | 2 +- lambdas/slack_post/slackio.py | 7 ++++--- 6 files changed, 21 insertions(+), 13 deletions(-) diff --git a/cdk/stacks/pipeline_stack.py b/cdk/stacks/pipeline_stack.py index aa38888..88765e1 100644 --- a/cdk/stacks/pipeline_stack.py +++ b/cdk/stacks/pipeline_stack.py @@ -274,6 +274,9 @@ class PipelineStack(Stack): # Parquet to S3 — it never touches the catalog (Phase 3). self.exports_bucket.grant_read(self.classifier_fn, "raw/*") self.exports_bucket.grant_read_write(self.classifier_fn, "analytics/*") + # summary.json/details.json live under meta/ (kept out of the Athena + # table's analytics/ prefix so queries don't read JSON as Parquet). + self.exports_bucket.grant_read_write(self.classifier_fn, "meta/*") secretsmanager.Secret.from_secret_name_v2( self, "AnthropicKey", ANTHROPIC_SECRET ).grant_read(self.classifier_fn) @@ -335,7 +338,8 @@ class PipelineStack(Stack): ), ), ) - self.exports_bucket.grant_read(fn, "analytics/*") + # Slack Lambdas read only the daily JSON under meta/ (not the Parquet). + self.exports_bucket.grant_read(fn, "meta/*") slack_secret.grant_read(fn) dashboard_param.grant_read(fn) return fn diff --git a/grafana/dashboards/apm-work-orders.json b/grafana/dashboards/apm-work-orders.json index ffecd14..4a9f09f 100644 --- a/grafana/dashboards/apm-work-orders.json +++ b/grafana/dashboards/apm-work-orders.json @@ -179,7 +179,7 @@ "panels": [ { "id": 1, - "type": "bar-chart", + "type": "barchart", "title": "Category Distribution", "description": "Count of work orders by classification category for the selected snapshot date. Sorted descending by volume. Filter using the template variables above.", "gridPos": { "x": 0, "y": 0, "w": 14, "h": 9 }, @@ -366,7 +366,7 @@ }, { "id": 4, - "type": "bar-chart", + "type": "barchart", "title": "Escalations by Site", "description": "Total escalations per site for the selected snapshot date. Includes all escalation categories.", "gridPos": { "x": 8, "y": 9, "w": 16, "h": 8 }, diff --git a/lambdas/classifier/handler.py b/lambdas/classifier/handler.py index f356cac..f678c33 100644 --- a/lambdas/classifier/handler.py +++ b/lambdas/classifier/handler.py @@ -30,7 +30,8 @@ import pandas as pd import classify as clf -ANALYTICS_PREFIX = "analytics" +ANALYTICS_PREFIX = "analytics" # Parquet snapshots — the Glue/Athena table reads this +META_PREFIX = "meta" # summary.json/details.json — kept OUT of the table's prefix # Map snapshot field -> substring matched (case-insensitively) against the export # header, tolerating minor header drift in the 13-column APM export. @@ -237,17 +238,19 @@ def handler(event, context): mode="overwrite_partitions", ) + # summary.json/details.json go under meta/ — NOT analytics/. Athena reads + # every object in the table's prefix as Parquet, so JSON there breaks queries. summary = _build_summary(df, dt, key, blank) _s3.put_object( Bucket=bucket, - Key=f"{ANALYTICS_PREFIX}/dt={dt}/summary.json", + Key=f"{META_PREFIX}/dt={dt}/summary.json", Body=json.dumps(summary, indent=2).encode("utf-8"), ContentType="application/json", ) # details.json — per-WO index the slack-post Lambda reads for drill-down modals. _s3.put_object( Bucket=bucket, - Key=f"{ANALYTICS_PREFIX}/dt={dt}/details.json", + Key=f"{META_PREFIX}/dt={dt}/details.json", Body=json.dumps(_build_details(df)).encode("utf-8"), ContentType="application/json", ) diff --git a/lambdas/slack_post/handler.py b/lambdas/slack_post/handler.py index 128b851..b0c54bb 100644 --- a/lambdas/slack_post/handler.py +++ b/lambdas/slack_post/handler.py @@ -25,11 +25,11 @@ def handler(event, context): """Post the daily summary and (conditionally) the 3rd-escalation alert.""" dt = (event or {}).get("dt") or datetime.now(timezone.utc).strftime("%Y-%m-%d") - today = slackio.read_analytics_json(dt, "summary.json") + today = slackio.read_meta_json(dt, "summary.json") if today is None: print(f"No summary.json for dt={dt}; nothing to post.") return {"posted": False, "reason": "no summary", "dt": dt} - yesterday = slackio.read_analytics_json(_yesterday(dt), "summary.json") + yesterday = slackio.read_meta_json(_yesterday(dt), "summary.json") client = slackio.web_client() channel = slackio.channel_id() @@ -44,7 +44,7 @@ def handler(event, context): # Standalone batched alert — only when there are 3rd escalations. Pull the # rows from details.json so the alert can name the WOs. if today.get("third_escalation_count", 0) > 0: - details = slackio.read_analytics_json(dt, "details.json") or [] + details = slackio.read_meta_json(dt, "details.json") or [] thirds = [d for d in details if d.get("category") == "3rd Escalation"] alert_blocks = blockkit.build_escalation_alert(thirds) if alert_blocks: diff --git a/lambdas/slack_post/interactions.py b/lambdas/slack_post/interactions.py index b664fda..5df79f3 100644 --- a/lambdas/slack_post/interactions.py +++ b/lambdas/slack_post/interactions.py @@ -60,7 +60,7 @@ def handler(event, context): # The daily post is same-day; default the drill to today's snapshot. dt = datetime.now(timezone.utc).strftime("%Y-%m-%d") - details = slackio.read_analytics_json(dt, "details.json") or [] + details = slackio.read_meta_json(dt, "details.json") or [] wos = [d for d in details if d.get(field) == value] view = blockkit.build_wo_modal( diff --git a/lambdas/slack_post/slackio.py b/lambdas/slack_post/slackio.py index 6b4b670..6d0a102 100644 --- a/lambdas/slack_post/slackio.py +++ b/lambdas/slack_post/slackio.py @@ -60,9 +60,10 @@ def verify_signature(body: str, timestamp: str, signature: str) -> bool: return verifier.is_valid(body=body, timestamp=timestamp, signature=signature) -def read_analytics_json(dt: str, name: str): - """Read ``analytics/dt=
/`` as JSON, or None if absent.""" - key = f"analytics/dt={dt}/{name}" +def read_meta_json(dt: str, name: str): + """Read ``meta/dt=
/`` as JSON, or None if absent. (summary.json / + details.json live under meta/, kept out of the Athena table's analytics/ prefix.)""" + key = f"meta/dt={dt}/{name}" try: obj = _s3.get_object(Bucket=_BUCKET, Key=key) except _s3.exceptions.NoSuchKey: From 3c8b7704f66c618644a8544e9962bd29952388b4 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 18:59:09 -0400 Subject: [PATCH 06/12] Fix Grafana Athena auth: use default credential chain, not ec2_iam_role MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Grafana rejected the datasource with 'trying to use non-allowed auth method ec2_iam_role: Failed to create client' — the plugin's allowed_auth_providers defaults to default,keys,credentials and excludes ec2_iam_role. Switch authType to 'default' (AWS SDK default chain), which on EC2 resolves to the instance role via IMDS (still no static keys) and is allowed out of the box. --- grafana/provisioning/datasources/athena.yaml | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/grafana/provisioning/datasources/athena.yaml b/grafana/provisioning/datasources/athena.yaml index 58072bf..bd955d9 100644 --- a/grafana/provisioning/datasources/athena.yaml +++ b/grafana/provisioning/datasources/athena.yaml @@ -1,5 +1,8 @@ # Athena datasource, authenticated via the EC2 instance IAM role (no static keys). # Provisioned into /etc/grafana/provisioning/datasources/ via user-data (Phase 5). +# authType "default" = AWS SDK default credential chain, which on EC2 resolves to +# the instance role via IMDS. ("ec2_iam_role" is rejected by the plugin unless +# added to [aws] allowed_auth_providers; "default" is allowed out of the box.) apiVersion: 1 datasources: - name: Athena @@ -7,7 +10,7 @@ datasources: uid: athena isDefault: true jsonData: - authType: ec2_iam_role + authType: default defaultRegion: us-east-1 catalog: AwsDataCatalog database: apm_wo_analysis From dce53fdc864599d29a715c65b4e0f78975fb9303 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 19:01:24 -0400 Subject: [PATCH 07/12] Fix Grafana template vars: refresh on dashboard load, not time-range change MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit All 6 query variables had refresh=2 (on time-range change) with no cached value, so a plain dashboard load never populated them — $dt resolved to empty and every panel filtered WHERE dt='' (no data). Set refresh=1 (on dashboard load). --- grafana/dashboards/apm-work-orders.json | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/grafana/dashboards/apm-work-orders.json b/grafana/dashboards/apm-work-orders.json index 4a9f09f..314a951 100644 --- a/grafana/dashboards/apm-work-orders.json +++ b/grafana/dashboards/apm-work-orders.json @@ -31,7 +31,7 @@ "region": "us-east-1" } }, - "refresh": 2, + "refresh": 1, "sort": 0, "multi": false, "includeAll": false, @@ -57,7 +57,7 @@ "region": "us-east-1" } }, - "refresh": 2, + "refresh": 1, "sort": 1, "multi": true, "includeAll": true, @@ -84,7 +84,7 @@ "region": "us-east-1" } }, - "refresh": 2, + "refresh": 1, "sort": 1, "multi": true, "includeAll": true, @@ -111,7 +111,7 @@ "region": "us-east-1" } }, - "refresh": 2, + "refresh": 1, "sort": 1, "multi": true, "includeAll": true, @@ -138,7 +138,7 @@ "region": "us-east-1" } }, - "refresh": 2, + "refresh": 1, "sort": 1, "multi": true, "includeAll": true, @@ -165,7 +165,7 @@ "region": "us-east-1" } }, - "refresh": 2, + "refresh": 1, "sort": 1, "multi": true, "includeAll": true, From d884de8fcc3323d696700b8a676b8b962bd1cf48 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Thu, 28 May 2026 19:06:06 -0400 Subject: [PATCH 08/12] Fix Grafana config-sync: --exact-timestamps for same-size updates aws s3 sync skips same-size files on download unless --exact-timestamps is set, so a dashboard edit that doesn't change file size (e.g. refresh 2->1, or a query tweak) never propagated to the instance. Add --exact-timestamps to all four sync invocations (boot + 15-min timer). --- cdk/assets/grafana_userdata.sh | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/cdk/assets/grafana_userdata.sh b/cdk/assets/grafana_userdata.sh index f67ade6..8720329 100644 --- a/cdk/assets/grafana_userdata.sh +++ b/cdk/assets/grafana_userdata.sh @@ -52,8 +52,8 @@ INI # --- sync provisioning + dashboards from S3 (repo is source of truth) --- sync_config() { - aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/provisioning/" /etc/grafana/provisioning/ --delete - aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/dashboards/" /var/lib/grafana/dashboards/ --delete + aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/provisioning/" /etc/grafana/provisioning/ --delete --exact-timestamps + aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/dashboards/" /var/lib/grafana/dashboards/ --delete --exact-timestamps chown -R grafana:grafana /etc/grafana/provisioning /var/lib/grafana/dashboards } mkdir -p /var/lib/grafana/dashboards @@ -66,8 +66,8 @@ systemctl enable --now grafana-server cat >/usr/local/bin/grafana-config-sync.sh < Date: Fri, 29 May 2026 10:55:02 -0400 Subject: [PATCH 09/12] Fix Grafana panels: rawSQL not rawSql (Athena plugin query key) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit All 13 panel/variable queries keyed the SQL as rawSql (lowercase); the grafana-athena-datasource plugin reads rawSQL (capital SQL). With the wrong key the plugin saw an empty query, so no Athena query ever fired — variables had no options and every panel showed a clean 'No data' (no error). This was the root cause of the empty dashboard; data/datasource/permissions were all fine. --- grafana/dashboards/apm-work-orders.json | 26 ++++++++++++------------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/grafana/dashboards/apm-work-orders.json b/grafana/dashboards/apm-work-orders.json index 314a951..afb69c9 100644 --- a/grafana/dashboards/apm-work-orders.json +++ b/grafana/dashboards/apm-work-orders.json @@ -23,7 +23,7 @@ "uid": "athena" }, "query": { - "rawSql": "SELECT DISTINCT dt FROM apm_wo_analysis.apm_wo_snapshots ORDER BY dt DESC", + "rawSQL": "SELECT DISTINCT dt FROM apm_wo_analysis.apm_wo_snapshots ORDER BY dt DESC", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -49,7 +49,7 @@ "uid": "athena" }, "query": { - "rawSql": "SELECT DISTINCT site FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND site<>'' ORDER BY 1", + "rawSQL": "SELECT DISTINCT site FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND site<>'' ORDER BY 1", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -76,7 +76,7 @@ "uid": "athena" }, "query": { - "rawSql": "SELECT DISTINCT department FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND department<>'' ORDER BY 1", + "rawSQL": "SELECT DISTINCT department FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND department<>'' ORDER BY 1", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -103,7 +103,7 @@ "uid": "athena" }, "query": { - "rawSql": "SELECT DISTINCT category FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND category<>'' ORDER BY 1", + "rawSQL": "SELECT DISTINCT category FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND category<>'' ORDER BY 1", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -130,7 +130,7 @@ "uid": "athena" }, "query": { - "rawSql": "SELECT DISTINCT wo_status FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND wo_status<>'' ORDER BY 1", + "rawSQL": "SELECT DISTINCT wo_status FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND wo_status<>'' ORDER BY 1", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -157,7 +157,7 @@ "uid": "athena" }, "query": { - "rawSql": "SELECT DISTINCT hold_reason FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND hold_reason<>'' ORDER BY 1", + "rawSQL": "SELECT DISTINCT hold_reason FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND hold_reason<>'' ORDER BY 1", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -194,7 +194,7 @@ "type": "grafana-athena-datasource", "uid": "athena" }, - "rawSql": "SELECT category, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1 ORDER BY 2 DESC", + "rawSQL": "SELECT category, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1 ORDER BY 2 DESC", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -257,7 +257,7 @@ "type": "grafana-athena-datasource", "uid": "athena" }, - "rawSql": "SELECT category, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true GROUP BY 1 ORDER BY 2 DESC", + "rawSQL": "SELECT category, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true GROUP BY 1 ORDER BY 2 DESC", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -325,7 +325,7 @@ "type": "grafana-athena-datasource", "uid": "athena" }, - "rawSql": "SELECT CASE WHEN is_action=true THEN 'Action Needed' ELSE 'Routine' END AS action_type, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1", + "rawSQL": "SELECT CASE WHEN is_action=true THEN 'Action Needed' ELSE 'Routine' END AS action_type, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -381,7 +381,7 @@ "type": "grafana-athena-datasource", "uid": "athena" }, - "rawSql": "SELECT site, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true AND site<>'' GROUP BY 1 ORDER BY 2 DESC", + "rawSQL": "SELECT site, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true AND site<>'' GROUP BY 1 ORDER BY 2 DESC", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -445,7 +445,7 @@ "type": "grafana-athena-datasource", "uid": "athena" }, - "rawSql": "SELECT date_parse(dt, '%Y-%m-%d') AS time, SUM(CAST(is_escalation AS INTEGER)) AS escalations, SUM(CAST(is_action AS INTEGER)) AS action_needed FROM apm_wo_analysis.apm_wo_snapshots GROUP BY 1 ORDER BY 1", + "rawSQL": "SELECT date_parse(dt, '%Y-%m-%d') AS time, SUM(CAST(is_escalation AS INTEGER)) AS escalations, SUM(CAST(is_action AS INTEGER)) AS action_needed FROM apm_wo_analysis.apm_wo_snapshots GROUP BY 1 ORDER BY 1", "format": "timeSeries", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -526,7 +526,7 @@ "type": "grafana-athena-datasource", "uid": "athena" }, - "rawSql": "SELECT wo_number, site, department, category, wo_status, hold_reason, wo_description, last_comment FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND ('${site:raw}' = 'All' OR site IN (${site:singlequote})) AND ('${department:raw}' = 'All' OR department IN (${department:singlequote})) AND ('${category:raw}' = 'All' OR category IN (${category:singlequote})) AND ('${wo_status:raw}' = 'All' OR wo_status IN (${wo_status:singlequote})) AND ('${hold_reason:raw}' = 'All' OR hold_reason IN (${hold_reason:singlequote})) ORDER BY category, site", + "rawSQL": "SELECT wo_number, site, department, category, wo_status, hold_reason, wo_description, last_comment FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND ('${site:raw}' = 'All' OR site IN (${site:singlequote})) AND ('${department:raw}' = 'All' OR department IN (${department:singlequote})) AND ('${category:raw}' = 'All' OR category IN (${category:singlequote})) AND ('${wo_status:raw}' = 'All' OR wo_status IN (${wo_status:singlequote})) AND ('${hold_reason:raw}' = 'All' OR hold_reason IN (${hold_reason:singlequote})) ORDER BY category, site", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", @@ -681,7 +681,7 @@ "type": "grafana-athena-datasource", "uid": "athena" }, - "rawSql": "SELECT wo_number, site, category, wo_status, hold_reason, mismatch FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND mismatch<>'' ORDER BY category, site", + "rawSQL": "SELECT wo_number, site, category, wo_status, hold_reason, mismatch FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND mismatch<>'' ORDER BY category, site", "format": "table", "connectionArgs": { "catalog": "AwsDataCatalog", From c970635be10823f5bfdc93d99b5e42994ff9d94e Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Fri, 29 May 2026 11:13:10 -0400 Subject: [PATCH 10/12] Fix WO table filter: drop custom allValue so :singlequote expands All The 5 multi-select filter vars had allValue='All'. Grafana does NOT apply the :singlequote format to a custom allValue, so 'All' was injected bare into site IN (ALL) -> Athena read ALL as a column ('Column ALL cannot be resolved'). Removing the custom allValue lets :singlequote expand the All selection to the real quoted value list, so the IN clause is valid SQL. --- grafana/dashboards/apm-work-orders.json | 5 ----- 1 file changed, 5 deletions(-) diff --git a/grafana/dashboards/apm-work-orders.json b/grafana/dashboards/apm-work-orders.json index afb69c9..528792d 100644 --- a/grafana/dashboards/apm-work-orders.json +++ b/grafana/dashboards/apm-work-orders.json @@ -61,7 +61,6 @@ "sort": 1, "multi": true, "includeAll": true, - "allValue": "All", "current": {}, "options": [], "hide": 0 @@ -88,7 +87,6 @@ "sort": 1, "multi": true, "includeAll": true, - "allValue": "All", "current": {}, "options": [], "hide": 0 @@ -115,7 +113,6 @@ "sort": 1, "multi": true, "includeAll": true, - "allValue": "All", "current": {}, "options": [], "hide": 0 @@ -142,7 +139,6 @@ "sort": 1, "multi": true, "includeAll": true, - "allValue": "All", "current": {}, "options": [], "hide": 0 @@ -169,7 +165,6 @@ "sort": 1, "multi": true, "includeAll": true, - "allValue": "All", "current": {}, "options": [], "hide": 0 From eae4d67e0037c2159c21b984506152abfa048c96 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Fri, 29 May 2026 11:29:14 -0400 Subject: [PATCH 11/12] Apply cross-review findings (Phase 2/4/5 hardening) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit From the cross_reviewer (GPT-4.1) per-PR passes, now that the orchestrator is back up: Phase 2 (classifier): - Process ALL S3 records, not just event["Records"][0] — batched notifications no longer silently dropped (the review's only BLOCK). - Derive the partition dt from the S3 event time, not the Lambda wall-clock — stable across retries / the midnight boundary. - Add an SQS dead-letter queue so a failed run surfaces instead of dropping a day's data after Lambda's retries. Phase 4 (Slack): - Stage throttling (rate 10 / burst 20) on the public /slack/interactions HTTP API. (AWS WAF doesn't attach to apigwv2 HTTP APIs; stage throttling is the mechanism.) Phase 5 (Grafana): - Explicit encrypted=True on the gp3 root volume. Tests: synth assertions for the DLQ, stage throttling, and the encrypted volume. 60/60 pass; cdk synth green for both stacks. Deferred NITs (print->logging, sig- failure source-IP logging, S3 versioning, CIDR-maintenance runbook) -> Phase 6. NOTE: like the earlier deploy fixes these sit on phase-5 but span phases — the classifier/DLQ to #8, throttling to #10, encryption to #11 — reconcile at merge. The encrypted-volume change needs the deferred clean instance replacement to take effect (can't encrypt a live volume in place). --- cdk/stacks/grafana_stack.py | 3 +- cdk/stacks/pipeline_stack.py | 22 +++++++++++++++ lambdas/classifier/handler.py | 53 ++++++++++++++++++++++++----------- tests/test_grafana_synth.py | 6 +++- tests/test_pipeline_synth.py | 30 ++++++++++++++++++++ 5 files changed, 95 insertions(+), 19 deletions(-) diff --git a/cdk/stacks/grafana_stack.py b/cdk/stacks/grafana_stack.py index ee1a713..17ee09d 100644 --- a/cdk/stacks/grafana_stack.py +++ b/cdk/stacks/grafana_stack.py @@ -188,11 +188,12 @@ class GrafanaStack(Stack): block_devices=[ ec2.BlockDevice( device_name="/dev/xvda", - # RETAIN the gp3 root volume (grafana.db lives here). + # RETAIN the gp3 root volume (grafana.db lives here), encrypted. volume=ec2.BlockDeviceVolume.ebs( 20, volume_type=ec2.EbsDeviceVolumeType.GP3, delete_on_termination=False, + encrypted=True, ), ) ], diff --git a/cdk/stacks/pipeline_stack.py b/cdk/stacks/pipeline_stack.py index 88765e1..89bbe53 100644 --- a/cdk/stacks/pipeline_stack.py +++ b/cdk/stacks/pipeline_stack.py @@ -56,6 +56,9 @@ from aws_cdk import ( from aws_cdk import ( aws_secretsmanager as secretsmanager, ) +from aws_cdk import ( + aws_sqs as sqs, +) from aws_cdk import ( aws_ssm as ssm, ) @@ -225,6 +228,16 @@ class PipelineStack(Stack): retention=logs.RetentionDays.TWO_MONTHS, removal_policy=RemovalPolicy.DESTROY, ) + # DLQ for failed async invocations — a daily pipeline must surface a + # failed run (malformed export, transient error) rather than silently + # drop a day's data after Lambda's retries. + classifier_dlq = sqs.Queue( + self, + "ClassifierDlq", + queue_name="apm-wo-analysis-classifier-dlq", + retention_period=Duration.days(14), + enforce_ssl=True, + ) self.classifier_fn = lambda_.Function( self, "Classifier", @@ -235,6 +248,7 @@ class PipelineStack(Stack): memory_size=512, timeout=Duration.seconds(120), log_group=classifier_logs, + dead_letter_queue=classifier_dlq, environment={"APM_HAIKU_FALLBACK": "on"}, # awswrangler/pandas/pyarrow/numpy come from the AWS-managed # SDK-for-pandas layer (pre-stripped to fit the 250 MB unzipped @@ -383,6 +397,14 @@ class PipelineStack(Stack): "InteractionsIntegration", interactions_fn ), ) + # Stage throttling on the public endpoint — Slack interactions are + # low-volume, so cap rate/burst to blunt abuse/DoS against this + # internet-facing route. (AWS WAF doesn't attach to HTTP APIs; stage + # throttling is the apigwv2 mechanism.) Escape hatch to the default stage. + default_stage = slack_api.default_stage.node.default_child + default_stage.default_route_settings = apigwv2.CfnStage.RouteSettingsProperty( + throttling_rate_limit=10, throttling_burst_limit=20 + ) # Route53 alias apm-wo.seahaven.com → the API Gateway custom domain. zone = route53.HostedZone.from_hosted_zone_attributes( diff --git a/lambdas/classifier/handler.py b/lambdas/classifier/handler.py index f678c33..36cc5de 100644 --- a/lambdas/classifier/handler.py +++ b/lambdas/classifier/handler.py @@ -202,19 +202,22 @@ def _build_details(df: pd.DataFrame) -> list[dict]: ] -def handler(event, context): - """Classify each export dropped under raw/ into a daily Parquet snapshot.""" - record = event["Records"][0] - bucket = record["s3"]["bucket"]["name"] - key = unquote_plus(record["s3"]["object"]["key"]) +def _event_dt(record: dict) -> str: + """Partition date from the S3 event time, not the Lambda wall-clock — stable + across retries and across a midnight boundary (a late-night upload retried + after midnight keeps the upload day's partition).""" + ts = record.get("eventTime") # ISO-8601, e.g. "2026-05-28T22:23:40.123Z" + return ts[:10] if ts else datetime.now(timezone.utc).strftime("%Y-%m-%d") + +def _process_object(bucket: str, key: str, dt: str) -> dict | None: + """Classify one export into the dt partition + write its summary/details. + Returns the summary dict, or None if the object isn't a usable export.""" if not key.startswith("raw/") or not key.lower().endswith((".xlsx", ".csv")): print(f"Skipping non-export object: s3://{bucket}/{key}") - return {"skipped": key} + return None - dt = datetime.now(timezone.utc).strftime("%Y-%m-%d") print(f"Classifying s3://{bucket}/{key} into dt={dt}") - with tempfile.NamedTemporaryFile(suffix=os.path.splitext(key)[1]) as tmp: _s3.download_fileobj(bucket, key, tmp) tmp.flush() @@ -222,8 +225,8 @@ def handler(event, context): df, blank = _build_snapshot(header, data) if df.empty: - print("No classifiable rows (all comments blank); nothing written.") - return {"classified": 0, "blank": blank} + print(f"No classifiable rows in {key} (all comments blank); nothing written.") + return None df["dt"] = dt # Pure Parquet write — no Glue registration. The apm_wo_snapshots table is @@ -254,17 +257,33 @@ def handler(event, context): Body=json.dumps(_build_details(df)).encode("utf-8"), ContentType="application/json", ) - print( f"Wrote {len(df)} rows, {summary['escalation_total']} escalations " f"({summary['third_escalation_count']} 3rd), {len(summary['mismatches'])} mismatches." ) - _invoke_slack_post(dt) - return { - "classified": int(len(df)), - "dt": dt, - "summary_key": f"dt={dt}/summary.json", - } + return summary + + +def handler(event, context): + """Classify EVERY export in the S3 event (S3 can batch multiple records), + then trigger the Slack post once per affected day. Raises on any failure so + the event is retried / lands in the DLQ rather than being silently dropped.""" + processed: list[dict] = [] + dts: set[str] = set() + for record in event.get("Records", []): + bucket = record["s3"]["bucket"]["name"] + key = unquote_plus(record["s3"]["object"]["key"]) + dt = _event_dt(record) + summary = _process_object(bucket, key, dt) + if summary is not None: + processed.append( + {"key": key, "dt": dt, "classified": summary["classified_total"]} + ) + dts.add(dt) + + for dt in sorted(dts): + _invoke_slack_post(dt) + return {"processed": processed} def _invoke_slack_post(dt: str) -> None: diff --git a/tests/test_grafana_synth.py b/tests/test_grafana_synth.py index 42f29b1..b6784c6 100644 --- a/tests/test_grafana_synth.py +++ b/tests/test_grafana_synth.py @@ -147,7 +147,11 @@ def test_root_volume_gp3_and_retained(): Match.object_like( { "Ebs": Match.object_like( - {"VolumeType": "gp3", "DeleteOnTermination": False} + { + "VolumeType": "gp3", + "DeleteOnTermination": False, + "Encrypted": True, + } ) } ) diff --git a/tests/test_pipeline_synth.py b/tests/test_pipeline_synth.py index eec10ac..641201c 100644 --- a/tests/test_pipeline_synth.py +++ b/tests/test_pipeline_synth.py @@ -150,6 +150,36 @@ def test_slack_lambdas_exist(): ) +def test_classifier_has_dlq(): + # Failed async invocations must surface, not silently drop a day's data. + t = _template() + t.resource_count_is("AWS::SQS::Queue", 1) + t.has_resource_properties( + "AWS::Lambda::Function", + Match.object_like( + { + "FunctionName": "apm-wo-analysis-classifier", + "DeadLetterConfig": Match.any_value(), + } + ), + ) + + +def test_interactions_stage_is_throttled(): + # The public Slack interactions endpoint caps rate/burst. + _template().has_resource_properties( + "AWS::ApiGatewayV2::Stage", + Match.object_like( + { + "DefaultRouteSettings": { + "ThrottlingRateLimit": 10, + "ThrottlingBurstLimit": 20, + } + } + ), + ) + + def test_interactions_api_routes_post_to_slack_endpoint(): t = _template() t.resource_count_is("AWS::ApiGatewayV2::Api", 1) From 0ca738ca8c18e5afd5914d110ea8466daa3e2f13 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Fri, 29 May 2026 13:36:56 -0400 Subject: [PATCH 12/12] Enable QEMU for arm64 Lambda bundling in CI/CD --- .github/workflows/ci.yaml | 1 + .github/workflows/deploy.yaml | 1 + 2 files changed, 2 insertions(+) diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index a30b1ad..f8c2130 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -13,3 +13,4 @@ jobs: run-cdk-synth: true cdk-dir: cdk run-tests: false + enable-qemu: true diff --git a/.github/workflows/deploy.yaml b/.github/workflows/deploy.yaml index d7cbfef..158fcca 100644 --- a/.github/workflows/deploy.yaml +++ b/.github/workflows/deploy.yaml @@ -18,5 +18,6 @@ jobs: python-version: "3.12" region: us-east-1 cdk-dir: cdk + enable-qemu: true secrets: deploy-role-arn: ${{ secrets.AWS_DEPLOY_ROLE_ARN }}