This commit is contained in:
Adam Moussa 2026-05-29 17:37:04 +00:00 • committed by GitHub
commit fc93e332c8
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
16 changed files with 1516 additions and 52 deletions

View file

@ -142,7 +142,7 @@ cdk deploy apm-wo-analysis-grafana # EC2, ALB, SG, Route53, datasource role
## Status
**Phase 4 — Slack surfaces (in review).** Build-out proceeds per
**Phase 5 — Grafana (in review); build complete, docs remain.** Build-out per
[`docs/BUILD.md`](./docs/BUILD.md): ingestion → classifier → Glue/Athena → Slack
→ Grafana → docs.
@ -159,4 +159,10 @@ cdk deploy apm-wo-analysis-grafana # EC2, ALB, SG, Route53, datasource role
3rd-escalation alert, drill-down modals on `apm-wo.seahaven.com`) — implemented
and `cdk synth`-green; PR #10 open (stacked on Phase 3). App manifest in
`slack/manifest.yaml`.
- **Phases 5–6** (Grafana, final docs) — not started.
- **Phase 5** self-hosted Grafana (EC2 + ALB on `grafana.seahaven.com`,
office-IP-restricted, 7-panel dashboard as code) — implemented and
`cdk synth`-green; PR #11 open (stacked on Phase 4).
- **Phase 6** (final docs / Confluence / runbook) — not started.
All phase PRs are stacked (#6→#7→#8→#9→#10→#11) and **unmerged**; nothing is
deployed yet. Cross-review and `/security-review` are outstanding across the stack.

View file

@ -0,0 +1,94 @@
#!/bin/bash
# Grafana OSS bootstrap for the apm-wo-analysis dashboard host (Amazon Linux 2023,
# ARM64). Idempotent enough to re-run. Config + dashboards are pulled from S3
# (the repo is the source of truth); a systemd timer re-syncs dashboards so panel
# updates ship by re-uploading to S3 — no instance rebuild.
#
# Templated by CDK: __CONFIG_BUCKET__ / __CONFIG_PREFIX__ / __PLUGIN_VERSION__.
set -euxo pipefail
CONFIG_BUCKET="__CONFIG_BUCKET__"
CONFIG_PREFIX="__CONFIG_PREFIX__"
PLUGIN_VERSION="__PLUGIN_VERSION__"
# --- Grafana OSS repo + install ---
cat >/etc/yum.repos.d/grafana.repo <<'REPO'
[grafana]
name=grafana
baseurl=https://rpm.grafana.com
repo_gpgcheck=1
enabled=1
gpgcheck=1
gpgkey=https://rpm.grafana.com/gpg.key
sslverify=1
REPO
dnf install -y grafana
# --- Athena datasource plugin (pinned for reproducibility) ---
# --homepath is required or grafana-cli can't find its config defaults.
grafana-cli --homepath=/usr/share/grafana --pluginsDir=/var/lib/grafana/plugins \
plugins install grafana-athena-datasource "${PLUGIN_VERSION}"
# --- grafana.ini: behind the ALB at grafana.seahaven.com, kiosk-friendly ---
cat >/etc/grafana/grafana.ini <<'INI'
[server]
protocol = http
http_port = 3000
root_url = https://grafana.seahaven.com/
enforce_domain = false
[security]
# Behind an office-IP-restricted ALB; allow embedding for the kiosk wall display.
allow_embedding = true
cookie_secure = true
[users]
default_theme = dark
[analytics]
reporting_enabled = false
check_for_updates = false
INI
# --- sync provisioning + dashboards from S3 (repo is source of truth) ---
sync_config() {
aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/provisioning/" /etc/grafana/provisioning/ --delete --exact-timestamps
aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/dashboards/" /var/lib/grafana/dashboards/ --delete --exact-timestamps
chown -R grafana:grafana /etc/grafana/provisioning /var/lib/grafana/dashboards
}
mkdir -p /var/lib/grafana/dashboards
sync_config
systemctl daemon-reload
systemctl enable --now grafana-server
# --- systemd timer: re-sync dashboards every 15 min so repo edits land without a rebuild ---
cat >/usr/local/bin/grafana-config-sync.sh <<SYNC
#!/bin/bash
set -euo pipefail
aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/provisioning/" /etc/grafana/provisioning/ --delete --exact-timestamps
aws s3 sync "s3://${CONFIG_BUCKET}/${CONFIG_PREFIX}/dashboards/" /var/lib/grafana/dashboards/ --delete --exact-timestamps
chown -R grafana:grafana /etc/grafana/provisioning /var/lib/grafana/dashboards
SYNC
chmod +x /usr/local/bin/grafana-config-sync.sh
cat >/etc/systemd/system/grafana-config-sync.service <<'SVC'
[Unit]
Description=Sync apm-wo Grafana config/dashboards from S3
[Service]
Type=oneshot
ExecStart=/usr/local/bin/grafana-config-sync.sh
SVC
cat >/etc/systemd/system/grafana-config-sync.timer <<'TIMER'
[Unit]
Description=Periodic apm-wo Grafana config sync
[Timer]
OnBootSec=5min
OnUnitActiveSec=15min
[Install]
WantedBy=timers.target
TIMER
systemctl daemon-reload
systemctl enable --now grafana-config-sync.timer

View file

@ -5,6 +5,13 @@
"wildcardCertArn": "arn:aws:acm:us-east-1:328440206208:certificate/a66c0994-90d4-410a-a1d9-5595c2a3fae3",
"hostedZoneId": "Z06652411XKH89KTZD3XA",
"hostedZoneName": "seahaven.com",
"slackInteractionsDomain": "apm-wo.seahaven.com"
"slackInteractionsDomain": "apm-wo.seahaven.com",
"grafanaVpcId": "vpc-0d3d4b67bd0cf8a68",
"grafanaAzs": ["us-east-1a", "us-east-1b"],
"grafanaPublicSubnetIds": ["subnet-0eea820effe1b3ae5", "subnet-0012f5895182c1580"],
"grafanaPrivateSubnetIds": ["subnet-04e38c507e96f1926", "subnet-0a0b4fc6f296dfba5"],
"grafanaDomain": "grafana.seahaven.com",
"officeCidrs": ["47.21.61.4/32", "96.250.164.146/32"],
"athenaPluginVersion": "3.2.0"
}
}

View file

@ -1,22 +1,291 @@
"""Grafana stack: VPC import, EC2, ALB, SG, Route53, Athena datasource role.
"""Grafana stack: imported VPC, EC2 (Grafana OSS), ALB, SG, Route53, IAM, backup.
Scaffold — resources are added in Phase 5 of docs/BUILD.md. The running box is
the only non-serverless piece here (self-hosted Grafana OSS on a t4g.small,
ARM64, VPN-only) and carries an OS/Grafana patching + config-backup obligation.
Dashboards are provisioned as code from grafana/ — the running instance is never
the source of truth.
The only non-serverless piece in this repo (self-hosted Grafana OSS on a
t4g.small, ARM64, AL2023). Reachability: an internet-facing ALB whose security
group only admits the office CIDRs — there is no Client VPN in the account, so
"VPN-only" is realized as office-IP restriction (the same pattern syslog-server
uses). The instance sits in private subnets, reachable only from the ALB SG and
administered via SSM Session Manager (no SSH, no key pair).
Dashboards/datasources are provisioned as code: CDK uploads ``grafana/`` to an S3
config prefix; instance user-data syncs it on boot and a systemd timer re-syncs,
so the running box is never the source of truth. The gp3 root volume is RETAINed
and snapshotted daily by DLM; ``grafana.db`` lives there.
Config (cdk.json context): grafanaVpcId, grafanaAzs, grafana{Public,Private}SubnetIds,
grafanaDomain, officeCidrs, wildcardCertArn, hostedZoneId/Name, athenaPluginVersion.
"""
from aws_cdk import Stack
import os
from aws_cdk import (
CfnTag,
Stack,
Tags,
)
from aws_cdk import (
aws_certificatemanager as acm,
)
from aws_cdk import (
aws_dlm as dlm,
)
from aws_cdk import (
aws_ec2 as ec2,
)
from aws_cdk import (
aws_elasticloadbalancingv2 as elbv2,
)
from aws_cdk import (
aws_elasticloadbalancingv2_targets as elbv2_targets,
)
from aws_cdk import (
aws_iam as iam,
)
from aws_cdk import (
aws_route53 as route53,
)
from aws_cdk import (
aws_route53_targets as route53_targets,
)
from aws_cdk import (
aws_s3 as s3,
)
from aws_cdk import (
aws_s3_deployment as s3deploy,
)
from constructs import Construct
EXPORTS_BUCKET = "apm-wo-analysis-exports-328440206208"
CONFIG_PREFIX = "grafana-config"
BACKUP_TAG = "apm-grafana-backup"
GRAFANA_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "grafana")
USERDATA_PATH = os.path.join(
os.path.dirname(__file__), "..", "assets", "grafana_userdata.sh"
)
class GrafanaStack(Stack):
def __init__(self, scope: Construct, construct_id: str, **kwargs) -> None:
super().__init__(scope, construct_id, **kwargs)
# Phase 5 — EC2 (Amazon Linux 2023, Grafana OSS via user-data),
# internal ALB (HTTPS, *.seahaven.com ACM cert),
# SG ingress from VPN/office CIDRs only,
# Route53 alias grafana.seahaven.com,
# instance role: Athena + Glue + S3 read (no static keys). TODO
ctx = self.node.try_get_context
domain = ctx("grafanaDomain")
# Import the shared seahaven-vpc by explicit attributes (no context
# lookup, so the offline synth test needs no AWS credentials).
vpc = ec2.Vpc.from_vpc_attributes(
self,
"SeahavenVpc",
vpc_id=ctx("grafanaVpcId"),
availability_zones=ctx("grafanaAzs"),
public_subnet_ids=ctx("grafanaPublicSubnetIds"),
private_subnet_ids=ctx("grafanaPrivateSubnetIds"),
)
# ----- Security groups -----
alb_sg = ec2.SecurityGroup(
self,
"AlbSg",
vpc=vpc,
description="apm-wo grafana ALB",
allow_all_outbound=True,
)
for cidr in ctx("officeCidrs"):
alb_sg.add_ingress_rule(
ec2.Peer.ipv4(cidr), ec2.Port.tcp(443), f"HTTPS from office {cidr}"
)
instance_sg = ec2.SecurityGroup(
self,
"InstanceSg",
vpc=vpc,
description="apm-wo grafana instance",
allow_all_outbound=True,
)
instance_sg.add_ingress_rule(
alb_sg, ec2.Port.tcp(3000), "Grafana HTTP from the ALB only"
)
# ----- Instance role: Athena query + Glue read + S3 (no static keys) -----
role = iam.Role(
self,
"GrafanaInstanceRole",
assumed_by=iam.ServicePrincipal("ec2.amazonaws.com"),
managed_policies=[
# Session Manager admin access; no SSH / bastion / key pair.
iam.ManagedPolicy.from_aws_managed_policy_name(
"AmazonSSMManagedInstanceCore"
)
],
)
role.add_to_policy(
iam.PolicyStatement(
sid="AthenaQuery",
actions=[
"athena:StartQueryExecution",
"athena:StopQueryExecution",
"athena:GetQueryExecution",
"athena:GetQueryResults",
"athena:GetWorkGroup",
"athena:ListWorkGroups",
],
resources=[
f"arn:aws:athena:{self.region}:{self.account}:workgroup/apm-wo-analysis"
],
)
)
role.add_to_policy(
iam.PolicyStatement(
sid="GlueReadOnly",
actions=[
"glue:GetDatabase",
"glue:GetDatabases",
"glue:GetTable",
"glue:GetTables",
"glue:GetPartition",
"glue:GetPartitions",
],
resources=[
f"arn:aws:glue:{self.region}:{self.account}:catalog",
f"arn:aws:glue:{self.region}:{self.account}:database/apm_wo_analysis",
f"arn:aws:glue:{self.region}:{self.account}:table/apm_wo_analysis/*",
],
)
)
bucket = s3.Bucket.from_bucket_name(self, "Exports", EXPORTS_BUCKET)
# Read the analytics snapshots; read+write athena-results (query output).
bucket.grant_read(role, "analytics/*")
bucket.grant_read_write(role, "athena-results/*")
# Config sync reads the grafana-config prefix.
bucket.grant_read(role, f"{CONFIG_PREFIX}/*")
# ----- Instance (Grafana OSS via user-data) -----
with open(USERDATA_PATH, encoding="utf-8") as fh:
userdata_script = fh.read()
userdata_script = (
userdata_script.replace("__CONFIG_BUCKET__", EXPORTS_BUCKET)
.replace("__CONFIG_PREFIX__", CONFIG_PREFIX)
.replace("__PLUGIN_VERSION__", ctx("athenaPluginVersion"))
)
user_data = ec2.UserData.for_linux()
user_data.add_commands(userdata_script)
instance = ec2.Instance(
self,
"Grafana",
vpc=vpc,
vpc_subnets=ec2.SubnetSelection(
subnet_type=ec2.SubnetType.PRIVATE_WITH_EGRESS
),
instance_type=ec2.InstanceType("t4g.small"),
machine_image=ec2.MachineImage.latest_amazon_linux2023(
cpu_type=ec2.AmazonLinuxCpuType.ARM_64
),
security_group=instance_sg,
role=role,
user_data=user_data,
require_imdsv2=True,
block_devices=[
ec2.BlockDevice(
device_name="/dev/xvda",
# RETAIN the gp3 root volume (grafana.db lives here), encrypted.
volume=ec2.BlockDeviceVolume.ebs(
20,
volume_type=ec2.EbsDeviceVolumeType.GP3,
delete_on_termination=False,
encrypted=True,
),
)
],
)
Tags.of(instance).add(BACKUP_TAG, "true")
# ----- ALB (internet-facing, office-IP-restricted, HTTPS) -----
cert = acm.Certificate.from_certificate_arn(
self, "WildcardCert", ctx("wildcardCertArn")
)
alb = elbv2.ApplicationLoadBalancer(
self,
"Alb",
vpc=vpc,
internet_facing=True,
security_group=alb_sg,
vpc_subnets=ec2.SubnetSelection(subnet_type=ec2.SubnetType.PUBLIC),
)
listener = alb.add_listener(
"Https",
port=443,
protocol=elbv2.ApplicationProtocol.HTTPS,
certificates=[cert],
# Do NOT auto-open 0.0.0.0/0 on 443 — the ALB SG already scopes
# ingress to the office CIDRs. open=True would undo that.
open=False,
)
listener.add_targets(
"GrafanaTarget",
port=3000,
protocol=elbv2.ApplicationProtocol.HTTP,
targets=[elbv2_targets.InstanceTarget(instance, 3000)],
health_check=elbv2.HealthCheck(
path="/api/health", healthy_http_codes="200"
),
)
# ----- Route53 alias grafana.seahaven.com -> ALB -----
zone = route53.HostedZone.from_hosted_zone_attributes(
self,
"SeahavenZone",
hosted_zone_id=ctx("hostedZoneId"),
zone_name=ctx("hostedZoneName"),
)
route53.ARecord(
self,
"GrafanaAlias",
zone=zone,
record_name=domain.split(".")[0],
target=route53.RecordTarget.from_alias(
route53_targets.LoadBalancerTarget(alb)
),
)
# ----- Dashboards-as-code: upload grafana/ to the S3 config prefix -----
s3deploy.BucketDeployment(
self,
"GrafanaConfig",
sources=[s3deploy.Source.asset(GRAFANA_DIR)],
destination_bucket=bucket,
destination_key_prefix=CONFIG_PREFIX,
prune=True,
)
# ----- Daily DLM snapshot of the (tagged) instance's volume -----
dlm_role = iam.Role(
self,
"DlmRole",
assumed_by=iam.ServicePrincipal("dlm.amazonaws.com"),
managed_policies=[
iam.ManagedPolicy.from_aws_managed_policy_name(
"service-role/AWSDataLifecycleManagerServiceRole"
)
],
)
dlm.CfnLifecyclePolicy(
self,
"GrafanaBackup",
description="Daily snapshot of the apm-wo grafana volume",
state="ENABLED",
execution_role_arn=dlm_role.role_arn,
policy_details=dlm.CfnLifecyclePolicy.PolicyDetailsProperty(
resource_types=["INSTANCE"],
target_tags=[CfnTag(key=BACKUP_TAG, value="true")],
schedules=[
dlm.CfnLifecyclePolicy.ScheduleProperty(
name="daily",
create_rule=dlm.CfnLifecyclePolicy.CreateRuleProperty(
interval=24, interval_unit="HOURS", times=["07:00"]
),
retain_rule=dlm.CfnLifecyclePolicy.RetainRuleProperty(count=7),
)
],
),
)

View file

@ -56,6 +56,9 @@ from aws_cdk import (
from aws_cdk import (
aws_secretsmanager as secretsmanager,
)
from aws_cdk import (
aws_sqs as sqs,
)
from aws_cdk import (
aws_ssm as ssm,
)
@ -64,6 +67,11 @@ from constructs import Construct
GLUE_DATABASE = "apm_wo_analysis"
GLUE_TABLE = "apm_wo_snapshots"
ATHENA_WORKGROUP = "apm-wo-analysis"
# AWS-managed SDK-for-pandas layer (awswrangler 3.16.1, py3.12, arm64). Provides
# awswrangler/pandas/pyarrow/numpy pre-stripped to fit the Lambda size limit.
AWSSDKPANDAS_LAYER_ARN = (
"arn:aws:lambda:us-east-1:336392948345:layer:AWSSDKPandas-Python312-Arm64:27"
)
ANTHROPIC_SECRET = "apm-wo-analysis/anthropic-api-key"
SLACK_SECRET = "apm-wo-analysis/slack-credentials"
GRAFANA_URL_PARAM = "/apm-wo-analysis/grafana-base-url"
@ -220,6 +228,16 @@ class PipelineStack(Stack):
retention=logs.RetentionDays.TWO_MONTHS,
removal_policy=RemovalPolicy.DESTROY,
)
# DLQ for failed async invocations — a daily pipeline must surface a
# failed run (malformed export, transient error) rather than silently
# drop a day's data after Lambda's retries.
classifier_dlq = sqs.Queue(
self,
"ClassifierDlq",
queue_name="apm-wo-analysis-classifier-dlq",
retention_period=Duration.days(14),
enforce_ssl=True,
)
self.classifier_fn = lambda_.Function(
self,
"Classifier",
@ -230,7 +248,17 @@ class PipelineStack(Stack):
memory_size=512,
timeout=Duration.seconds(120),
log_group=classifier_logs,
dead_letter_queue=classifier_dlq,
environment={"APM_HAIKU_FALLBACK": "on"},
# awswrangler/pandas/pyarrow/numpy come from the AWS-managed
# SDK-for-pandas layer (pre-stripped to fit the 250 MB unzipped
# limit, which bundling them ourselves blows). The function package
# only bundles openpyxl; boto3 is in the runtime, urllib is stdlib.
layers=[
lambda_.LayerVersion.from_layer_version_arn(
self, "PandasLayer", AWSSDKPANDAS_LAYER_ARN
)
],
code=lambda_.Code.from_asset(
os.path.join(LAMBDAS_DIR, "classifier"),
bundling=BundlingOptions(
@ -260,6 +288,9 @@ class PipelineStack(Stack):
# Parquet to S3 — it never touches the catalog (Phase 3).
self.exports_bucket.grant_read(self.classifier_fn, "raw/*")
self.exports_bucket.grant_read_write(self.classifier_fn, "analytics/*")
# summary.json/details.json live under meta/ (kept out of the Athena
# table's analytics/ prefix so queries don't read JSON as Parquet).
self.exports_bucket.grant_read_write(self.classifier_fn, "meta/*")
secretsmanager.Secret.from_secret_name_v2(
self, "AnthropicKey", ANTHROPIC_SECRET
).grant_read(self.classifier_fn)
@ -321,7 +352,8 @@ class PipelineStack(Stack):
),
),
)
self.exports_bucket.grant_read(fn, "analytics/*")
# Slack Lambdas read only the daily JSON under meta/ (not the Parquet).
self.exports_bucket.grant_read(fn, "meta/*")
slack_secret.grant_read(fn)
dashboard_param.grant_read(fn)
return fn
@ -365,6 +397,14 @@ class PipelineStack(Stack):
"InteractionsIntegration", interactions_fn
),
)
# Stage throttling on the public endpoint — Slack interactions are
# low-volume, so cap rate/burst to blunt abuse/DoS against this
# internet-facing route. (AWS WAF doesn't attach to HTTP APIs; stage
# throttling is the apigwv2 mechanism.) Escape hatch to the default stage.
default_stage = slack_api.default_stage.node.default_child
default_stage.default_route_settings = apigwv2.CfnStage.RouteSettingsProperty(
throttling_rate_limit=10, throttling_burst_limit=20
)
# Route53 alias apm-wo.seahaven.com → the API Gateway custom domain.
zone = route53.HostedZone.from_hosted_zone_attributes(

View file

@ -1,7 +1,7 @@
{
"uid": "apm-wo",
"title": "APM Work Orders",
"description": "Daily APM work-order analysis — breakdown, escalations, trend, filterable WO table, mismatches. Built in Phase 5 (docs/BUILD.md) by the grafana-author agent.",
"description": "Daily APM work-order analysis — breakdown, escalations, trend, filterable WO table, mismatches. Built in Phase 5 by the grafana-author agent. Athena datasource uid='athena'; data grain is one row per WO per daily snapshot partitioned by dt.",
"tags": ["apm", "work-orders"],
"timezone": "browser",
"schemaVersion": 39,
@ -12,7 +12,794 @@
"to": "now"
},
"templating": {
"list": []
"list": [
{
"name": "dt",
"label": "Snapshot Date",
"description": "Single daily snapshot partition. Defaults to the latest available dt. Most panels filter WHERE dt = '$dt'.",
"type": "query",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"query": {
"rawSQL": "SELECT DISTINCT dt FROM apm_wo_analysis.apm_wo_snapshots ORDER BY dt DESC",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
},
"refresh": 1,
"sort": 0,
"multi": false,
"includeAll": false,
"current": {},
"options": [],
"hide": 0
},
{
"name": "site",
"label": "Site",
"description": "Multi-select. WO table applies: AND ('${site:raw}' = 'All' OR site IN (${site:singlequote}))",
"type": "query",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"query": {
"rawSQL": "SELECT DISTINCT site FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND site<>'' ORDER BY 1",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
},
"refresh": 1,
"sort": 1,
"multi": true,
"includeAll": true,
"current": {},
"options": [],
"hide": 0
},
{
"name": "department",
"label": "Department",
"description": "Multi-select. WO table applies: AND ('${department:raw}' = 'All' OR department IN (${department:singlequote}))",
"type": "query",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"query": {
"rawSQL": "SELECT DISTINCT department FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND department<>'' ORDER BY 1",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
},
"refresh": 1,
"sort": 1,
"multi": true,
"includeAll": true,
"current": {},
"options": [],
"hide": 0
},
{
"name": "category",
"label": "Category",
"description": "Multi-select. WO table applies: AND ('${category:raw}' = 'All' OR category IN (${category:singlequote}))",
"type": "query",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"query": {
"rawSQL": "SELECT DISTINCT category FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND category<>'' ORDER BY 1",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
},
"refresh": 1,
"sort": 1,
"multi": true,
"includeAll": true,
"current": {},
"options": [],
"hide": 0
},
{
"name": "wo_status",
"label": "WO Status",
"description": "Multi-select. WO table applies: AND ('${wo_status:raw}' = 'All' OR wo_status IN (${wo_status:singlequote}))",
"type": "query",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"query": {
"rawSQL": "SELECT DISTINCT wo_status FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND wo_status<>'' ORDER BY 1",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
},
"refresh": 1,
"sort": 1,
"multi": true,
"includeAll": true,
"current": {},
"options": [],
"hide": 0
},
{
"name": "hold_reason",
"label": "Hold Reason",
"description": "Multi-select. WO table applies: AND ('${hold_reason:raw}' = 'All' OR hold_reason IN (${hold_reason:singlequote}))",
"type": "query",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"query": {
"rawSQL": "SELECT DISTINCT hold_reason FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND hold_reason<>'' ORDER BY 1",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
},
"refresh": 1,
"sort": 1,
"multi": true,
"includeAll": true,
"current": {},
"options": [],
"hide": 0
}
]
},
"panels": []
"panels": [
{
"id": 1,
"type": "barchart",
"title": "Category Distribution",
"description": "Count of work orders by classification category for the selected snapshot date. Sorted descending by volume. Filter using the template variables above.",
"gridPos": { "x": 0, "y": 0, "w": 14, "h": 9 },
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"targets": [
{
"refId": "A",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"rawSQL": "SELECT category, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1 ORDER BY 2 DESC",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
}
],
"options": {
"orientation": "horizontal",
"barRadius": 0,
"groupWidth": 0.7,
"showValue": "always",
"stacking": "none",
"tooltip": { "mode": "single", "sort": "none" },
"legend": { "showLegend": false, "displayMode": "list", "placement": "bottom" }
},
"fieldConfig": {
"defaults": {
"color": { "mode": "palette-classic" },
"custom": {
"axisBorderShow": false,
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"fillOpacity": 80,
"gradientMode": "none",
"hideFrom": { "legend": false, "tooltip": false, "viz": false },
"lineWidth": 1,
"scaleDistribution": { "type": "linear" },
"thresholdsStyle": { "mode": "off" }
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "green", "value": null },
{ "color": "red", "value": 80 }
]
}
},
"overrides": []
}
},
{
"id": 2,
"type": "piechart",
"title": "Escalation Summary",
"description": "Distribution across escalation categories (1st, 2nd, 3rd Escalation, SIM Ticket, Other Escalation) for the selected snapshot date. 3rd Escalation is highlighted red.",
"gridPos": { "x": 14, "y": 0, "w": 10, "h": 9 },
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"targets": [
{
"refId": "A",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"rawSQL": "SELECT category, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true GROUP BY 1 ORDER BY 2 DESC",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
}
],
"options": {
"pieType": "pie",
"displayLabels": ["name", "value"],
"tooltip": { "mode": "single", "sort": "none" },
"legend": { "showLegend": true, "displayMode": "table", "placement": "right", "values": ["value", "percent"] }
},
"fieldConfig": {
"defaults": {
"color": { "mode": "palette-classic" },
"custom": {
"hideFrom": { "legend": false, "tooltip": false, "viz": false }
},
"mappings": []
},
"overrides": [
{
"matcher": { "id": "byName", "options": "3rd Escalation" },
"properties": [
{ "id": "color", "value": { "fixedColor": "red", "mode": "fixed" } }
]
},
{
"matcher": { "id": "byName", "options": "2nd Escalation" },
"properties": [
{ "id": "color", "value": { "fixedColor": "orange", "mode": "fixed" } }
]
},
{
"matcher": { "id": "byName", "options": "1st Escalation" },
"properties": [
{ "id": "color", "value": { "fixedColor": "yellow", "mode": "fixed" } }
]
},
{
"matcher": { "id": "byName", "options": "SIM Ticket" },
"properties": [
{ "id": "color", "value": { "fixedColor": "purple", "mode": "fixed" } }
]
}
]
}
},
{
"id": 3,
"type": "piechart",
"title": "Action-Needed vs Routine",
"description": "Action-needed WOs require attention (escalations, scheduling, vendor/report waits, status inquiries, vendor no-shows). Routine WOs are on track. Counts are for the selected snapshot date.",
"gridPos": { "x": 0, "y": 9, "w": 8, "h": 8 },
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"targets": [
{
"refId": "A",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"rawSQL": "SELECT CASE WHEN is_action=true THEN 'Action Needed' ELSE 'Routine' END AS action_type, COUNT(*) AS wos FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' GROUP BY 1",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
}
],
"options": {
"pieType": "donut",
"displayLabels": ["name", "percent"],
"tooltip": { "mode": "single", "sort": "none" },
"legend": { "showLegend": true, "displayMode": "list", "placement": "bottom", "values": ["value", "percent"] }
},
"fieldConfig": {
"defaults": {
"color": { "mode": "palette-classic" },
"custom": {
"hideFrom": { "legend": false, "tooltip": false, "viz": false }
},
"mappings": []
},
"overrides": [
{
"matcher": { "id": "byName", "options": "Action Needed" },
"properties": [
{ "id": "color", "value": { "fixedColor": "semi-dark-orange", "mode": "fixed" } }
]
},
{
"matcher": { "id": "byName", "options": "Routine" },
"properties": [
{ "id": "color", "value": { "fixedColor": "green", "mode": "fixed" } }
]
}
]
}
},
{
"id": 4,
"type": "barchart",
"title": "Escalations by Site",
"description": "Total escalations per site for the selected snapshot date. Includes all escalation categories.",
"gridPos": { "x": 8, "y": 9, "w": 16, "h": 8 },
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"targets": [
{
"refId": "A",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"rawSQL": "SELECT site, COUNT(*) AS escalations FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND is_escalation=true AND site<>'' GROUP BY 1 ORDER BY 2 DESC",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
}
],
"options": {
"orientation": "vertical",
"barRadius": 0,
"groupWidth": 0.7,
"showValue": "always",
"stacking": "none",
"tooltip": { "mode": "single", "sort": "none" },
"legend": { "showLegend": false, "displayMode": "list", "placement": "bottom" },
"xTickLabelRotation": -45,
"xTickLabelMaxLength": 12
},
"fieldConfig": {
"defaults": {
"color": { "mode": "fixed", "fixedColor": "semi-dark-red" },
"custom": {
"axisBorderShow": false,
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "Escalations",
"axisPlacement": "auto",
"fillOpacity": 80,
"gradientMode": "none",
"hideFrom": { "legend": false, "tooltip": false, "viz": false },
"lineWidth": 1,
"scaleDistribution": { "type": "linear" },
"thresholdsStyle": { "mode": "off" }
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "green", "value": null }
]
}
},
"overrides": []
}
},
{
"id": 5,
"type": "timeseries",
"title": "Trend Over Time — Escalations & Action-Needed per Day",
"description": "Runs across all partitions (no $dt filter) to show daily escalation and action-needed volume over time. Use the dashboard time range picker to zoom in. This is the capability the legacy Sheet never had.",
"gridPos": { "x": 0, "y": 17, "w": 24, "h": 9 },
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"targets": [
{
"refId": "A",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"rawSQL": "SELECT date_parse(dt, '%Y-%m-%d') AS time, SUM(CAST(is_escalation AS INTEGER)) AS escalations, SUM(CAST(is_action AS INTEGER)) AS action_needed FROM apm_wo_analysis.apm_wo_snapshots GROUP BY 1 ORDER BY 1",
"format": "timeSeries",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
}
],
"options": {
"tooltip": { "mode": "multi", "sort": "desc" },
"legend": { "showLegend": true, "displayMode": "list", "placement": "bottom", "calcs": ["mean", "max", "last"] }
},
"fieldConfig": {
"defaults": {
"color": { "mode": "palette-classic" },
"custom": {
"axisBorderShow": false,
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "Work Orders",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 10,
"gradientMode": "none",
"hideFrom": { "legend": false, "tooltip": false, "viz": false },
"insertNulls": false,
"lineInterpolation": "linear",
"lineWidth": 2,
"pointSize": 5,
"scaleDistribution": { "type": "linear" },
"showPoints": "auto",
"spanNulls": false,
"stacking": { "group": "A", "mode": "none" },
"thresholdsStyle": { "mode": "off" }
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "green", "value": null }
]
},
"unit": "short"
},
"overrides": [
{
"matcher": { "id": "byName", "options": "escalations" },
"properties": [
{ "id": "color", "value": { "fixedColor": "semi-dark-red", "mode": "fixed" } },
{ "id": "displayName", "value": "Escalations" }
]
},
{
"matcher": { "id": "byName", "options": "action_needed" },
"properties": [
{ "id": "color", "value": { "fixedColor": "semi-dark-orange", "mode": "fixed" } },
{ "id": "displayName", "value": "Action Needed" }
]
}
]
}
},
{
"id": 6,
"type": "table",
"title": "Work Order Detail Table",
"description": "Filterable, exportable table of all work orders for the selected snapshot date. Multi-value template variables are applied via: AND ('${variable:raw}' = 'All' OR column IN (${variable:singlequote})). Escalation rows are color-coded: 3rd Escalation = red, 2nd Escalation = orange, 1st Escalation = yellow. WO numbers are plain text (no APM deep-link per project decision). Use the Download CSV button (table header menu) to export. last_comment is wrapped for readability.",
"gridPos": { "x": 0, "y": 26, "w": 24, "h": 14 },
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"targets": [
{
"refId": "A",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"rawSQL": "SELECT wo_number, site, department, category, wo_status, hold_reason, wo_description, last_comment FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND ('${site:raw}' = 'All' OR site IN (${site:singlequote})) AND ('${department:raw}' = 'All' OR department IN (${department:singlequote})) AND ('${category:raw}' = 'All' OR category IN (${category:singlequote})) AND ('${wo_status:raw}' = 'All' OR wo_status IN (${wo_status:singlequote})) AND ('${hold_reason:raw}' = 'All' OR hold_reason IN (${hold_reason:singlequote})) ORDER BY category, site",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
}
],
"options": {
"frameIndex": 0,
"showHeader": true,
"sortBy": [],
"footer": {
"show": false,
"reducer": ["sum"],
"fields": "",
"enablePagination": false
}
},
"fieldConfig": {
"defaults": {
"color": { "mode": "thresholds" },
"custom": {
"align": "left",
"cellOptions": { "type": "auto" },
"inspect": false,
"filterable": true,
"minWidth": 80,
"width": 0
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "text", "value": null }
]
}
},
"overrides": [
{
"matcher": { "id": "byName", "options": "wo_number" },
"properties": [
{ "id": "displayName", "value": "WO #" },
{ "id": "custom.width", "value": 100 }
]
},
{
"matcher": { "id": "byName", "options": "site" },
"properties": [
{ "id": "displayName", "value": "Site" },
{ "id": "custom.width", "value": 80 }
]
},
{
"matcher": { "id": "byName", "options": "department" },
"properties": [
{ "id": "displayName", "value": "Dept" },
{ "id": "custom.width", "value": 80 }
]
},
{
"matcher": { "id": "byName", "options": "category" },
"properties": [
{ "id": "displayName", "value": "Category" },
{ "id": "custom.width", "value": 160 },
{
"id": "mappings",
"value": [
{
"type": "value",
"options": {
"3rd Escalation": {
"color": "dark-red",
"index": 0
},
"2nd Escalation": {
"color": "dark-orange",
"index": 1
},
"1st Escalation": {
"color": "dark-yellow",
"index": 2
},
"SIM Ticket": {
"color": "dark-purple",
"index": 3
},
"Other Escalation": {
"color": "orange",
"index": 4
}
}
}
]
},
{
"id": "custom.cellOptions",
"value": { "type": "color-text" }
}
]
},
{
"matcher": { "id": "byName", "options": "wo_status" },
"properties": [
{ "id": "displayName", "value": "Status" },
{ "id": "custom.width", "value": 80 }
]
},
{
"matcher": { "id": "byName", "options": "hold_reason" },
"properties": [
{ "id": "displayName", "value": "Hold Reason" },
{ "id": "custom.width", "value": 120 }
]
},
{
"matcher": { "id": "byName", "options": "wo_description" },
"properties": [
{ "id": "displayName", "value": "Description" },
{ "id": "custom.width", "value": 220 }
]
},
{
"matcher": { "id": "byName", "options": "last_comment" },
"properties": [
{ "id": "displayName", "value": "Last Comment" },
{
"id": "custom.cellOptions",
"value": { "type": "auto", "wrapText": true }
},
{ "id": "custom.width", "value": 420 },
{ "id": "custom.minWidth", "value": 200 }
]
}
]
}
},
{
"id": 7,
"type": "table",
"title": "Mismatch Panel — Comment vs Structured-State Contradictions",
"description": "Work orders where the classified comment intent contradicts the structured WO state (e.g. comment says 'completed' but WO is on a REPORT/VENDOR/SCHEDULING hold, or comment says 'scheduled' while status is IP). Non-empty mismatch column only. Use these rows for manual review before the next export.",
"gridPos": { "x": 0, "y": 40, "w": 24, "h": 8 },
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"targets": [
{
"refId": "A",
"datasource": {
"type": "grafana-athena-datasource",
"uid": "athena"
},
"rawSQL": "SELECT wo_number, site, category, wo_status, hold_reason, mismatch FROM apm_wo_analysis.apm_wo_snapshots WHERE dt='$dt' AND mismatch<>'' ORDER BY category, site",
"format": "table",
"connectionArgs": {
"catalog": "AwsDataCatalog",
"database": "apm_wo_analysis",
"region": "us-east-1"
}
}
],
"options": {
"frameIndex": 0,
"showHeader": true,
"sortBy": [],
"footer": {
"show": false,
"reducer": ["sum"],
"fields": "",
"enablePagination": false
}
},
"fieldConfig": {
"defaults": {
"color": { "mode": "thresholds" },
"custom": {
"align": "left",
"cellOptions": { "type": "auto" },
"inspect": false,
"filterable": true,
"minWidth": 80,
"width": 0
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "text", "value": null }
]
}
},
"overrides": [
{
"matcher": { "id": "byName", "options": "wo_number" },
"properties": [
{ "id": "displayName", "value": "WO #" },
{ "id": "custom.width", "value": 100 }
]
},
{
"matcher": { "id": "byName", "options": "site" },
"properties": [
{ "id": "displayName", "value": "Site" },
{ "id": "custom.width", "value": 80 }
]
},
{
"matcher": { "id": "byName", "options": "category" },
"properties": [
{ "id": "displayName", "value": "Category" },
{ "id": "custom.width", "value": 160 },
{
"id": "mappings",
"value": [
{
"type": "value",
"options": {
"3rd Escalation": {
"color": "dark-red",
"index": 0
},
"2nd Escalation": {
"color": "dark-orange",
"index": 1
},
"1st Escalation": {
"color": "dark-yellow",
"index": 2
},
"SIM Ticket": {
"color": "dark-purple",
"index": 3
},
"Other Escalation": {
"color": "orange",
"index": 4
}
}
}
]
},
{
"id": "custom.cellOptions",
"value": { "type": "color-text" }
}
]
},
{
"matcher": { "id": "byName", "options": "wo_status" },
"properties": [
{ "id": "displayName", "value": "WO Status" },
{ "id": "custom.width", "value": 90 }
]
},
{
"matcher": { "id": "byName", "options": "hold_reason" },
"properties": [
{ "id": "displayName", "value": "Hold Reason" },
{ "id": "custom.width", "value": 120 }
]
},
{
"matcher": { "id": "byName", "options": "mismatch" },
"properties": [
{ "id": "displayName", "value": "Mismatch Description" },
{
"id": "custom.cellOptions",
"value": { "type": "auto", "wrapText": true }
},
{ "id": "custom.width", "value": 500 },
{ "id": "custom.minWidth", "value": 200 },
{ "id": "color", "value": { "mode": "fixed", "fixedColor": "semi-dark-orange" } }
]
}
]
}
}
]
}

View file

@ -1,12 +1,16 @@
# Athena datasource, authenticated via the EC2 instance IAM role (no static keys).
# Provisioned into /etc/grafana/provisioning/datasources/ via user-data (Phase 5).
# authType "default" = AWS SDK default credential chain, which on EC2 resolves to
# the instance role via IMDS. ("ec2_iam_role" is rejected by the plugin unless
# added to [aws] allowed_auth_providers; "default" is allowed out of the box.)
apiVersion: 1
datasources:
- name: Athena
type: grafana-athena-datasource
uid: athena
isDefault: true
jsonData:
authType: ec2_iam_role
authType: default
defaultRegion: us-east-1
catalog: AwsDataCatalog
database: apm_wo_analysis

View file

@ -30,7 +30,8 @@ import pandas as pd
import classify as clf
ANALYTICS_PREFIX = "analytics"
ANALYTICS_PREFIX = "analytics" # Parquet snapshots — the Glue/Athena table reads this
META_PREFIX = "meta" # summary.json/details.json — kept OUT of the table's prefix
# Map snapshot field -> substring matched (case-insensitively) against the export
# header, tolerating minor header drift in the 13-column APM export.
@ -201,19 +202,22 @@ def _build_details(df: pd.DataFrame) -> list[dict]:
]
def handler(event, context):
"""Classify each export dropped under raw/ into a daily Parquet snapshot."""
record = event["Records"][0]
bucket = record["s3"]["bucket"]["name"]
key = unquote_plus(record["s3"]["object"]["key"])
def _event_dt(record: dict) -> str:
"""Partition date from the S3 event time, not the Lambda wall-clock — stable
across retries and across a midnight boundary (a late-night upload retried
after midnight keeps the upload day's partition)."""
ts = record.get("eventTime") # ISO-8601, e.g. "2026-05-28T22:23:40.123Z"
return ts[:10] if ts else datetime.now(timezone.utc).strftime("%Y-%m-%d")
def _process_object(bucket: str, key: str, dt: str) -> dict | None:
"""Classify one export into the dt partition + write its summary/details.
Returns the summary dict, or None if the object isn't a usable export."""
if not key.startswith("raw/") or not key.lower().endswith((".xlsx", ".csv")):
print(f"Skipping non-export object: s3://{bucket}/{key}")
return {"skipped": key}
return None
dt = datetime.now(timezone.utc).strftime("%Y-%m-%d")
print(f"Classifying s3://{bucket}/{key} into dt={dt}")
with tempfile.NamedTemporaryFile(suffix=os.path.splitext(key)[1]) as tmp:
_s3.download_fileobj(bucket, key, tmp)
tmp.flush()
@ -221,8 +225,8 @@ def handler(event, context):
df, blank = _build_snapshot(header, data)
if df.empty:
print("No classifiable rows (all comments blank); nothing written.")
return {"classified": 0, "blank": blank}
print(f"No classifiable rows in {key} (all comments blank); nothing written.")
return None
df["dt"] = dt
# Pure Parquet write — no Glue registration. The apm_wo_snapshots table is
@ -237,31 +241,49 @@ def handler(event, context):
mode="overwrite_partitions",
)
# summary.json/details.json go under meta/ — NOT analytics/. Athena reads
# every object in the table's prefix as Parquet, so JSON there breaks queries.
summary = _build_summary(df, dt, key, blank)
_s3.put_object(
Bucket=bucket,
Key=f"{ANALYTICS_PREFIX}/dt={dt}/summary.json",
Key=f"{META_PREFIX}/dt={dt}/summary.json",
Body=json.dumps(summary, indent=2).encode("utf-8"),
ContentType="application/json",
)
# details.json — per-WO index the slack-post Lambda reads for drill-down modals.
_s3.put_object(
Bucket=bucket,
Key=f"{ANALYTICS_PREFIX}/dt={dt}/details.json",
Key=f"{META_PREFIX}/dt={dt}/details.json",
Body=json.dumps(_build_details(df)).encode("utf-8"),
ContentType="application/json",
)
print(
f"Wrote {len(df)} rows, {summary['escalation_total']} escalations "
f"({summary['third_escalation_count']} 3rd), {len(summary['mismatches'])} mismatches."
)
_invoke_slack_post(dt)
return {
"classified": int(len(df)),
"dt": dt,
"summary_key": f"dt={dt}/summary.json",
}
return summary
def handler(event, context):
"""Classify EVERY export in the S3 event (S3 can batch multiple records),
then trigger the Slack post once per affected day. Raises on any failure so
the event is retried / lands in the DLQ rather than being silently dropped."""
processed: list[dict] = []
dts: set[str] = set()
for record in event.get("Records", []):
bucket = record["s3"]["bucket"]["name"]
key = unquote_plus(record["s3"]["object"]["key"])
dt = _event_dt(record)
summary = _process_object(bucket, key, dt)
if summary is not None:
processed.append(
{"key": key, "dt": dt, "classified": summary["classified_total"]}
)
dts.add(dt)
for dt in sorted(dts):
_invoke_slack_post(dt)
return {"processed": processed}
def _invoke_slack_post(dt: str) -> None:

View file

@ -1,3 +1,5 @@
awswrangler>=3.9.0
# awswrangler + pandas/pyarrow/numpy come from the AWS-managed SDK-for-pandas
# Lambda layer (see pipeline_stack.py) — bundling them here blows the 250 MB
# unzipped limit. The Haiku fallback uses stdlib urllib, so no anthropic SDK.
# Only openpyxl (xlsx parsing) is bundled into the function package.
openpyxl>=3.1.0
anthropic>=0.40.0

View file

@ -282,8 +282,11 @@ def build_daily_summary(
for cat, count in drill_cats:
short_label = cat.replace(" Escalation", " Esc.").replace("Awaiting ", "")
short_label = short_label[:20] # button text kept concise
# action_id must be UNIQUE per message (Slack rejects duplicates), so
# qualify it with the category; the interactions handler matches on the
# "drill_category:" prefix and reads the filter value from `value`.
button_elements.append(
_button(f"{short_label} ({count})", "drill_category", cat)
_button(f"{short_label} ({count})", f"drill_category:{cat}", cat)
)
# Dashboard link button always present (no action_id — url button).

View file

@ -25,11 +25,11 @@ def handler(event, context):
"""Post the daily summary and (conditionally) the 3rd-escalation alert."""
dt = (event or {}).get("dt") or datetime.now(timezone.utc).strftime("%Y-%m-%d")
today = slackio.read_analytics_json(dt, "summary.json")
today = slackio.read_meta_json(dt, "summary.json")
if today is None:
print(f"No summary.json for dt={dt}; nothing to post.")
return {"posted": False, "reason": "no summary", "dt": dt}
yesterday = slackio.read_analytics_json(_yesterday(dt), "summary.json")
yesterday = slackio.read_meta_json(_yesterday(dt), "summary.json")
client = slackio.web_client()
channel = slackio.channel_id()
@ -44,7 +44,7 @@ def handler(event, context):
# Standalone batched alert — only when there are 3rd escalations. Pull the
# rows from details.json so the alert can name the WOs.
if today.get("third_escalation_count", 0) > 0:
details = slackio.read_analytics_json(dt, "details.json") or []
details = slackio.read_meta_json(dt, "details.json") or []
thirds = [d for d in details if d.get("category") == "3rd Escalation"]
alert_blocks = blockkit.build_escalation_alert(thirds)
if alert_blocks:

View file

@ -49,7 +49,10 @@ def handler(event, context):
return {"statusCode": 200, "body": ""}
action = (payload.get("actions") or [{}])[0]
field = _FILTER_FIELD.get(action.get("action_id"))
# action_id is qualified for Slack uniqueness, e.g. "drill_category:Report /
# Docs Needed" — match on the prefix before ":"; the filter value is in `value`.
action_kind = (action.get("action_id") or "").split(":", 1)[0]
field = _FILTER_FIELD.get(action_kind)
value = action.get("value")
trigger_id = payload.get("trigger_id")
if not field or not value or not trigger_id:
@ -57,7 +60,7 @@ def handler(event, context):
# The daily post is same-day; default the drill to today's snapshot.
dt = datetime.now(timezone.utc).strftime("%Y-%m-%d")
details = slackio.read_analytics_json(dt, "details.json") or []
details = slackio.read_meta_json(dt, "details.json") or []
wos = [d for d in details if d.get(field) == value]
view = blockkit.build_wo_modal(

View file

@ -60,9 +60,10 @@ def verify_signature(body: str, timestamp: str, signature: str) -> bool:
return verifier.is_valid(body=body, timestamp=timestamp, signature=signature)
def read_analytics_json(dt: str, name: str):
"""Read ``analytics/dt=<dt>/<name>`` as JSON, or None if absent."""
key = f"analytics/dt={dt}/{name}"
def read_meta_json(dt: str, name: str):
"""Read ``meta/dt=<dt>/<name>`` as JSON, or None if absent. (summary.json /
details.json live under meta/, kept out of the Athena table's analytics/ prefix.)"""
key = f"meta/dt={dt}/{name}"
try:
obj = _s3.get_object(Bucket=_BUCKET, Key=key)
except _s3.exceptions.NoSuchKey:

View file

@ -151,6 +151,20 @@ class TestDailySummaryStructure:
blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL)
assert any(b.get("type") == "header" for b in blocks)
def test_action_ids_are_unique(self):
# Slack rejects a message with duplicate action_ids across its elements
# (regression: all drill buttons once shared action_id "drill_category").
blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL)
action_ids = [
el["action_id"]
for b in blocks
for el in b.get("elements", [])
if isinstance(el, dict) and "action_id" in el
]
assert len(action_ids) == len(set(action_ids)), (
f"duplicate action_id(s): {action_ids}"
)
def test_header_contains_date(self):
blocks = build_daily_summary(TODAY_SUMMARY, YESTERDAY_SUMMARY, DASHBOARD_URL)
header = next(b for b in blocks if b.get("type") == "header")
@ -191,7 +205,8 @@ class TestDailySummaryStructure:
if block.get("type") != "actions":
continue
for elem in block.get("elements", []):
if elem.get("action_id") == "drill_category":
# action_id is qualified for uniqueness: "drill_category:<cat>".
if str(elem.get("action_id", "")).startswith("drill_category:"):
drill_found = True
assert "value" in elem
assert drill_found, "expected at least one drill_category action button"

181
tests/test_grafana_synth.py Normal file
View file

@ -0,0 +1,181 @@
"""Synth-level assertions for the Grafana stack (Phase 5).
Synthesizes ``apm-wo-analysis-grafana`` and asserts the security posture that
can't be eyeballed: the ALB only admits the office CIDRs on 443 (never
0.0.0.0/0), the instance only takes traffic from the ALB SG, the instance role
carries no static keys and only scoped Athena/Glue-read/S3 access, the root
volume is gp3 + retained, a daily DLM backup exists, and grafana.seahaven.com
aliases the ALB. No AWS, no Docker (bundling skipped).
Run with the repo venv:
python -m pytest tests/test_grafana_synth.py -q
"""
import json
import sys
from pathlib import Path
import aws_cdk as cdk
from aws_cdk.assertions import Match, Template
CDK_DIR = Path(__file__).resolve().parents[1] / "cdk"
sys.path.insert(0, str(CDK_DIR))
from stacks.grafana_stack import GrafanaStack # noqa: E402
OFFICE_CIDRS = {"47.21.61.4/32", "96.250.164.146/32"}
def _cdk_context() -> dict:
ctx = json.loads((CDK_DIR / "cdk.json").read_text())["context"]
ctx["aws:cdk:bundling-stacks"] = []
return ctx
def _template() -> Template:
app = cdk.App(context=_cdk_context())
stack = GrafanaStack(
app,
"apm-wo-analysis-grafana",
env=cdk.Environment(account="328440206208", region="us-east-1"),
)
return Template.from_stack(stack)
def _all_cidr_ingress(t: Template):
"""Every CIDR-based ingress rule, inline on SGs and standalone, as
(cidr, from_port, to_port) tuples."""
rules = []
for sg in t.find_resources("AWS::EC2::SecurityGroup").values():
for r in sg["Properties"].get("SecurityGroupIngress", []):
if "CidrIp" in r:
rules.append((r["CidrIp"], r.get("FromPort"), r.get("ToPort")))
for ing in t.find_resources("AWS::EC2::SecurityGroupIngress").values():
p = ing["Properties"]
if "CidrIp" in p:
rules.append((p["CidrIp"], p.get("FromPort"), p.get("ToPort")))
return rules
def test_alb_only_admits_office_cidrs_on_443():
rules = _all_cidr_ingress(_template())
cidrs_443 = {c for c, fp, tp in rules if fp == 443 and tp == 443}
assert cidrs_443 == OFFICE_CIDRS, (
f"443 ingress should be office-only, got {cidrs_443}"
)
# Nothing anywhere may be open to the world.
assert all(c != "0.0.0.0/0" for c, _, _ in rules), "found a 0.0.0.0/0 ingress"
def test_instance_only_reachable_from_alb_on_3000():
# The instance SG ingress on 3000 is a SourceSecurityGroup rule, not a CIDR.
_template().has_resource_properties(
"AWS::EC2::SecurityGroupIngress",
Match.object_like(
{
"FromPort": 3000,
"ToPort": 3000,
"SourceSecurityGroupId": Match.any_value(),
}
),
)
def test_alb_internet_facing_https_listener():
t = _template()
t.has_resource_properties(
"AWS::ElasticLoadBalancingV2::LoadBalancer", {"Scheme": "internet-facing"}
)
t.has_resource_properties(
"AWS::ElasticLoadBalancingV2::Listener",
Match.object_like(
{"Port": 443, "Protocol": "HTTPS", "Certificates": Match.any_value()}
),
)
def test_no_static_keys_in_stack():
t = _template()
t.resource_count_is("AWS::IAM::User", 0)
t.resource_count_is("AWS::IAM::AccessKey", 0)
def test_instance_role_scoped_and_uses_ssm():
t = _template()
# Session Manager (no SSH) — the SSM managed policy is attached.
t.has_resource_properties(
"AWS::IAM::Role",
Match.object_like(
{
"ManagedPolicyArns": Match.array_with(
[
{
"Fn::Join": [
"",
Match.array_with(
[":iam::aws:policy/AmazonSSMManagedInstanceCore"]
),
]
}
]
)
}
),
)
# The instance role must not be able to write the catalog or run wide Athena.
for policy in t.find_resources("AWS::IAM::Policy").values():
for stmt in policy["Properties"]["PolicyDocument"]["Statement"]:
actions = stmt.get("Action", [])
actions = [actions] if isinstance(actions, str) else actions
for a in actions:
if isinstance(a, str):
assert a not in ("glue:*", "athena:*", "s3:*", "*"), (
f"too broad: {a}"
)
assert not a.startswith("glue:Create"), f"no Glue writes: {a}"
assert not a.startswith("glue:Update"), f"no Glue writes: {a}"
def test_root_volume_gp3_and_retained():
_template().has_resource_properties(
"AWS::EC2::Instance",
Match.object_like(
{
"BlockDeviceMappings": Match.array_with(
[
Match.object_like(
{
"Ebs": Match.object_like(
{
"VolumeType": "gp3",
"DeleteOnTermination": False,
"Encrypted": True,
}
)
}
)
]
)
}
),
)
def test_daily_dlm_backup_enabled():
_template().has_resource_properties(
"AWS::DLM::LifecyclePolicy",
Match.object_like(
{
"State": "ENABLED",
"PolicyDetails": Match.object_like({"ResourceTypes": ["INSTANCE"]}),
}
),
)
def test_route53_alias_for_grafana():
_template().has_resource_properties(
"AWS::Route53::RecordSet",
Match.object_like({"Type": "A", "Name": "grafana.seahaven.com."}),
)

View file

@ -150,6 +150,36 @@ def test_slack_lambdas_exist():
)
def test_classifier_has_dlq():
# Failed async invocations must surface, not silently drop a day's data.
t = _template()
t.resource_count_is("AWS::SQS::Queue", 1)
t.has_resource_properties(
"AWS::Lambda::Function",
Match.object_like(
{
"FunctionName": "apm-wo-analysis-classifier",
"DeadLetterConfig": Match.any_value(),
}
),
)
def test_interactions_stage_is_throttled():
# The public Slack interactions endpoint caps rate/burst.
_template().has_resource_properties(
"AWS::ApiGatewayV2::Stage",
Match.object_like(
{
"DefaultRouteSettings": {
"ThrottlingRateLimit": 10,
"ThrottlingBurstLimit": 20,
}
}
),
)
def test_interactions_api_routes_post_to_slack_endpoint():
t = _template()
t.resource_count_is("AWS::ApiGatewayV2::Api", 1)