Add self-hosted Grafana stack: EC2, ALB, dashboards-as-code (Phase 5)
The one non-serverless piece — Grafana OSS on a t4g.small (AL2023, ARM64) in the
imported seahaven-vpc, fronted by an internet-facing ALB locked by SG to the
office CIDRs (no Client VPN exists, so "VPN-only" = office-IP restriction, the
syslog-server pattern). Instance in private subnets, reachable only from the ALB
SG, administered via SSM Session Manager (no SSH/key pair).
grafana_stack.py: ALB (HTTPS, *.seahaven.com cert, open=False so the SG office
rules aren't undone by an auto 0.0.0.0/0), instance role (Athena query + Glue
read + S3 analytics/athena-results, no static keys), Route53 grafana.seahaven.com
alias, gp3 root volume RETAINed, daily DLM snapshot of the tagged instance, and a
BucketDeployment that uploads grafana/ to the S3 config prefix.
grafana_userdata.sh: install Grafana OSS, pin the Athena datasource plugin, write
grafana.ini (root_url grafana.seahaven.com, kiosk embedding), sync provisioning +
dashboards from S3 on boot, and a systemd timer re-syncs every 15 min so repo
edits land without an instance rebuild.
Dashboard (grafana-author agent, grafana/dashboards/apm-work-orders.json, uid
apm-wo so the Slack 📊 button resolves): 7 panels — category distribution,
escalation summary, action/routine, escalations-by-site, trend time-series over
dt (the new capability), filterable WO table (5 template vars, escalation row
coloring, CSV export, no APM links), and the mismatch panel. Datasource uid
"athena" pinned in the provisioning yaml.
Tests: tests/test_grafana_synth.py — ALB admits only the office CIDRs on 443
(caught and fixed a default 0.0.0.0/0 listener rule), instance only-from-ALB,
no static keys, scoped instance role + SSM, gp3+retained root volume, daily DLM
backup, grafana.seahaven.com alias. 57/57 tests pass; full cdk synth green.
2026-05-28 18:05:20 -04:00
|
|
|
"""Synth-level assertions for the Grafana stack (Phase 5).
|
|
|
|
|
|
|
|
|
|
Synthesizes ``apm-wo-analysis-grafana`` and asserts the security posture that
|
|
|
|
|
can't be eyeballed: the ALB only admits the office CIDRs on 443 (never
|
|
|
|
|
0.0.0.0/0), the instance only takes traffic from the ALB SG, the instance role
|
|
|
|
|
carries no static keys and only scoped Athena/Glue-read/S3 access, the root
|
|
|
|
|
volume is gp3 + retained, a daily DLM backup exists, and grafana.seahaven.com
|
|
|
|
|
aliases the ALB. No AWS, no Docker (bundling skipped).
|
|
|
|
|
|
|
|
|
|
Run with the repo venv:
|
|
|
|
|
|
|
|
|
|
python -m pytest tests/test_grafana_synth.py -q
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
import json
|
|
|
|
|
import sys
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
import aws_cdk as cdk
|
|
|
|
|
from aws_cdk.assertions import Match, Template
|
|
|
|
|
|
|
|
|
|
CDK_DIR = Path(__file__).resolve().parents[1] / "cdk"
|
|
|
|
|
sys.path.insert(0, str(CDK_DIR))
|
|
|
|
|
|
|
|
|
|
from stacks.grafana_stack import GrafanaStack # noqa: E402
|
|
|
|
|
|
|
|
|
|
OFFICE_CIDRS = {"47.21.61.4/32", "96.250.164.146/32"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _cdk_context() -> dict:
|
|
|
|
|
ctx = json.loads((CDK_DIR / "cdk.json").read_text())["context"]
|
|
|
|
|
ctx["aws:cdk:bundling-stacks"] = []
|
|
|
|
|
return ctx
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _template() -> Template:
|
|
|
|
|
app = cdk.App(context=_cdk_context())
|
|
|
|
|
stack = GrafanaStack(
|
|
|
|
|
app,
|
|
|
|
|
"apm-wo-analysis-grafana",
|
|
|
|
|
env=cdk.Environment(account="328440206208", region="us-east-1"),
|
|
|
|
|
)
|
|
|
|
|
return Template.from_stack(stack)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _all_cidr_ingress(t: Template):
|
|
|
|
|
"""Every CIDR-based ingress rule, inline on SGs and standalone, as
|
|
|
|
|
(cidr, from_port, to_port) tuples."""
|
|
|
|
|
rules = []
|
|
|
|
|
for sg in t.find_resources("AWS::EC2::SecurityGroup").values():
|
|
|
|
|
for r in sg["Properties"].get("SecurityGroupIngress", []):
|
|
|
|
|
if "CidrIp" in r:
|
|
|
|
|
rules.append((r["CidrIp"], r.get("FromPort"), r.get("ToPort")))
|
|
|
|
|
for ing in t.find_resources("AWS::EC2::SecurityGroupIngress").values():
|
|
|
|
|
p = ing["Properties"]
|
|
|
|
|
if "CidrIp" in p:
|
|
|
|
|
rules.append((p["CidrIp"], p.get("FromPort"), p.get("ToPort")))
|
|
|
|
|
return rules
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_alb_only_admits_office_cidrs_on_443():
|
|
|
|
|
rules = _all_cidr_ingress(_template())
|
|
|
|
|
cidrs_443 = {c for c, fp, tp in rules if fp == 443 and tp == 443}
|
|
|
|
|
assert cidrs_443 == OFFICE_CIDRS, (
|
|
|
|
|
f"443 ingress should be office-only, got {cidrs_443}"
|
|
|
|
|
)
|
|
|
|
|
# Nothing anywhere may be open to the world.
|
|
|
|
|
assert all(c != "0.0.0.0/0" for c, _, _ in rules), "found a 0.0.0.0/0 ingress"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_instance_only_reachable_from_alb_on_3000():
|
|
|
|
|
# The instance SG ingress on 3000 is a SourceSecurityGroup rule, not a CIDR.
|
|
|
|
|
_template().has_resource_properties(
|
|
|
|
|
"AWS::EC2::SecurityGroupIngress",
|
|
|
|
|
Match.object_like(
|
|
|
|
|
{
|
|
|
|
|
"FromPort": 3000,
|
|
|
|
|
"ToPort": 3000,
|
|
|
|
|
"SourceSecurityGroupId": Match.any_value(),
|
|
|
|
|
}
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_alb_internet_facing_https_listener():
|
|
|
|
|
t = _template()
|
|
|
|
|
t.has_resource_properties(
|
|
|
|
|
"AWS::ElasticLoadBalancingV2::LoadBalancer", {"Scheme": "internet-facing"}
|
|
|
|
|
)
|
|
|
|
|
t.has_resource_properties(
|
|
|
|
|
"AWS::ElasticLoadBalancingV2::Listener",
|
|
|
|
|
Match.object_like(
|
|
|
|
|
{"Port": 443, "Protocol": "HTTPS", "Certificates": Match.any_value()}
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_no_static_keys_in_stack():
|
|
|
|
|
t = _template()
|
|
|
|
|
t.resource_count_is("AWS::IAM::User", 0)
|
|
|
|
|
t.resource_count_is("AWS::IAM::AccessKey", 0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_instance_role_scoped_and_uses_ssm():
|
|
|
|
|
t = _template()
|
|
|
|
|
# Session Manager (no SSH) — the SSM managed policy is attached.
|
|
|
|
|
t.has_resource_properties(
|
|
|
|
|
"AWS::IAM::Role",
|
|
|
|
|
Match.object_like(
|
|
|
|
|
{
|
|
|
|
|
"ManagedPolicyArns": Match.array_with(
|
|
|
|
|
[
|
|
|
|
|
{
|
|
|
|
|
"Fn::Join": [
|
|
|
|
|
"",
|
|
|
|
|
Match.array_with(
|
|
|
|
|
[":iam::aws:policy/AmazonSSMManagedInstanceCore"]
|
|
|
|
|
),
|
|
|
|
|
]
|
|
|
|
|
}
|
|
|
|
|
]
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
# The instance role must not be able to write the catalog or run wide Athena.
|
|
|
|
|
for policy in t.find_resources("AWS::IAM::Policy").values():
|
|
|
|
|
for stmt in policy["Properties"]["PolicyDocument"]["Statement"]:
|
|
|
|
|
actions = stmt.get("Action", [])
|
|
|
|
|
actions = [actions] if isinstance(actions, str) else actions
|
|
|
|
|
for a in actions:
|
|
|
|
|
if isinstance(a, str):
|
|
|
|
|
assert a not in ("glue:*", "athena:*", "s3:*", "*"), (
|
|
|
|
|
f"too broad: {a}"
|
|
|
|
|
)
|
|
|
|
|
assert not a.startswith("glue:Create"), f"no Glue writes: {a}"
|
|
|
|
|
assert not a.startswith("glue:Update"), f"no Glue writes: {a}"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_root_volume_gp3_and_retained():
|
|
|
|
|
_template().has_resource_properties(
|
|
|
|
|
"AWS::EC2::Instance",
|
|
|
|
|
Match.object_like(
|
|
|
|
|
{
|
|
|
|
|
"BlockDeviceMappings": Match.array_with(
|
|
|
|
|
[
|
|
|
|
|
Match.object_like(
|
|
|
|
|
{
|
|
|
|
|
"Ebs": Match.object_like(
|
Apply cross-review findings (Phase 2/4/5 hardening)
From the cross_reviewer (GPT-4.1) per-PR passes, now that the orchestrator is
back up:
Phase 2 (classifier):
- Process ALL S3 records, not just event["Records"][0] — batched notifications
no longer silently dropped (the review's only BLOCK).
- Derive the partition dt from the S3 event time, not the Lambda wall-clock —
stable across retries / the midnight boundary.
- Add an SQS dead-letter queue so a failed run surfaces instead of dropping a
day's data after Lambda's retries.
Phase 4 (Slack):
- Stage throttling (rate 10 / burst 20) on the public /slack/interactions HTTP
API. (AWS WAF doesn't attach to apigwv2 HTTP APIs; stage throttling is the
mechanism.)
Phase 5 (Grafana):
- Explicit encrypted=True on the gp3 root volume.
Tests: synth assertions for the DLQ, stage throttling, and the encrypted volume.
60/60 pass; cdk synth green for both stacks. Deferred NITs (print->logging, sig-
failure source-IP logging, S3 versioning, CIDR-maintenance runbook) -> Phase 6.
NOTE: like the earlier deploy fixes these sit on phase-5 but span phases — the
classifier/DLQ to #8, throttling to #10, encryption to #11 — reconcile at merge.
The encrypted-volume change needs the deferred clean instance replacement to
take effect (can't encrypt a live volume in place).
2026-05-29 11:29:14 -04:00
|
|
|
{
|
|
|
|
|
"VolumeType": "gp3",
|
|
|
|
|
"DeleteOnTermination": False,
|
|
|
|
|
"Encrypted": True,
|
|
|
|
|
}
|
Add self-hosted Grafana stack: EC2, ALB, dashboards-as-code (Phase 5)
The one non-serverless piece — Grafana OSS on a t4g.small (AL2023, ARM64) in the
imported seahaven-vpc, fronted by an internet-facing ALB locked by SG to the
office CIDRs (no Client VPN exists, so "VPN-only" = office-IP restriction, the
syslog-server pattern). Instance in private subnets, reachable only from the ALB
SG, administered via SSM Session Manager (no SSH/key pair).
grafana_stack.py: ALB (HTTPS, *.seahaven.com cert, open=False so the SG office
rules aren't undone by an auto 0.0.0.0/0), instance role (Athena query + Glue
read + S3 analytics/athena-results, no static keys), Route53 grafana.seahaven.com
alias, gp3 root volume RETAINed, daily DLM snapshot of the tagged instance, and a
BucketDeployment that uploads grafana/ to the S3 config prefix.
grafana_userdata.sh: install Grafana OSS, pin the Athena datasource plugin, write
grafana.ini (root_url grafana.seahaven.com, kiosk embedding), sync provisioning +
dashboards from S3 on boot, and a systemd timer re-syncs every 15 min so repo
edits land without an instance rebuild.
Dashboard (grafana-author agent, grafana/dashboards/apm-work-orders.json, uid
apm-wo so the Slack 📊 button resolves): 7 panels — category distribution,
escalation summary, action/routine, escalations-by-site, trend time-series over
dt (the new capability), filterable WO table (5 template vars, escalation row
coloring, CSV export, no APM links), and the mismatch panel. Datasource uid
"athena" pinned in the provisioning yaml.
Tests: tests/test_grafana_synth.py — ALB admits only the office CIDRs on 443
(caught and fixed a default 0.0.0.0/0 listener rule), instance only-from-ALB,
no static keys, scoped instance role + SSM, gp3+retained root volume, daily DLM
backup, grafana.seahaven.com alias. 57/57 tests pass; full cdk synth green.
2026-05-28 18:05:20 -04:00
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
)
|
|
|
|
|
]
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_daily_dlm_backup_enabled():
|
|
|
|
|
_template().has_resource_properties(
|
|
|
|
|
"AWS::DLM::LifecyclePolicy",
|
|
|
|
|
Match.object_like(
|
|
|
|
|
{
|
|
|
|
|
"State": "ENABLED",
|
|
|
|
|
"PolicyDetails": Match.object_like({"ResourceTypes": ["INSTANCE"]}),
|
|
|
|
|
}
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_route53_alias_for_grafana():
|
|
|
|
|
_template().has_resource_properties(
|
|
|
|
|
"AWS::Route53::RecordSet",
|
|
|
|
|
Match.object_like({"Type": "A", "Name": "grafana.seahaven.com."}),
|
|
|
|
|
)
|