apm-wo-analysis/cdk/stacks/grafana_stack.py
Adam Moussa eae4d67e00 Apply cross-review findings (Phase 2/4/5 hardening)
From the cross_reviewer (GPT-4.1) per-PR passes, now that the orchestrator is
back up:

Phase 2 (classifier):
- Process ALL S3 records, not just event["Records"][0] — batched notifications
  no longer silently dropped (the review's only BLOCK).
- Derive the partition dt from the S3 event time, not the Lambda wall-clock —
  stable across retries / the midnight boundary.
- Add an SQS dead-letter queue so a failed run surfaces instead of dropping a
  day's data after Lambda's retries.

Phase 4 (Slack):
- Stage throttling (rate 10 / burst 20) on the public /slack/interactions HTTP
  API. (AWS WAF doesn't attach to apigwv2 HTTP APIs; stage throttling is the
  mechanism.)

Phase 5 (Grafana):
- Explicit encrypted=True on the gp3 root volume.

Tests: synth assertions for the DLQ, stage throttling, and the encrypted volume.
60/60 pass; cdk synth green for both stacks. Deferred NITs (print->logging, sig-
failure source-IP logging, S3 versioning, CIDR-maintenance runbook) -> Phase 6.

NOTE: like the earlier deploy fixes these sit on phase-5 but span phases — the
classifier/DLQ to #8, throttling to #10, encryption to #11 — reconcile at merge.
The encrypted-volume change needs the deferred clean instance replacement to
take effect (can't encrypt a live volume in place).
2026-05-29 11:29:14 -04:00

291 lines
10 KiB
Python

"""Grafana stack: imported VPC, EC2 (Grafana OSS), ALB, SG, Route53, IAM, backup.
The only non-serverless piece in this repo (self-hosted Grafana OSS on a
t4g.small, ARM64, AL2023). Reachability: an internet-facing ALB whose security
group only admits the office CIDRs — there is no Client VPN in the account, so
"VPN-only" is realized as office-IP restriction (the same pattern syslog-server
uses). The instance sits in private subnets, reachable only from the ALB SG and
administered via SSM Session Manager (no SSH, no key pair).
Dashboards/datasources are provisioned as code: CDK uploads ``grafana/`` to an S3
config prefix; instance user-data syncs it on boot and a systemd timer re-syncs,
so the running box is never the source of truth. The gp3 root volume is RETAINed
and snapshotted daily by DLM; ``grafana.db`` lives there.
Config (cdk.json context): grafanaVpcId, grafanaAzs, grafana{Public,Private}SubnetIds,
grafanaDomain, officeCidrs, wildcardCertArn, hostedZoneId/Name, athenaPluginVersion.
"""
import os
from aws_cdk import (
CfnTag,
Stack,
Tags,
)
from aws_cdk import (
aws_certificatemanager as acm,
)
from aws_cdk import (
aws_dlm as dlm,
)
from aws_cdk import (
aws_ec2 as ec2,
)
from aws_cdk import (
aws_elasticloadbalancingv2 as elbv2,
)
from aws_cdk import (
aws_elasticloadbalancingv2_targets as elbv2_targets,
)
from aws_cdk import (
aws_iam as iam,
)
from aws_cdk import (
aws_route53 as route53,
)
from aws_cdk import (
aws_route53_targets as route53_targets,
)
from aws_cdk import (
aws_s3 as s3,
)
from aws_cdk import (
aws_s3_deployment as s3deploy,
)
from constructs import Construct
EXPORTS_BUCKET = "apm-wo-analysis-exports-328440206208"
CONFIG_PREFIX = "grafana-config"
BACKUP_TAG = "apm-grafana-backup"
GRAFANA_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "grafana")
USERDATA_PATH = os.path.join(
os.path.dirname(__file__), "..", "assets", "grafana_userdata.sh"
)
class GrafanaStack(Stack):
def __init__(self, scope: Construct, construct_id: str, **kwargs) -> None:
super().__init__(scope, construct_id, **kwargs)
ctx = self.node.try_get_context
domain = ctx("grafanaDomain")
# Import the shared seahaven-vpc by explicit attributes (no context
# lookup, so the offline synth test needs no AWS credentials).
vpc = ec2.Vpc.from_vpc_attributes(
self,
"SeahavenVpc",
vpc_id=ctx("grafanaVpcId"),
availability_zones=ctx("grafanaAzs"),
public_subnet_ids=ctx("grafanaPublicSubnetIds"),
private_subnet_ids=ctx("grafanaPrivateSubnetIds"),
)
# ----- Security groups -----
alb_sg = ec2.SecurityGroup(
self,
"AlbSg",
vpc=vpc,
description="apm-wo grafana ALB",
allow_all_outbound=True,
)
for cidr in ctx("officeCidrs"):
alb_sg.add_ingress_rule(
ec2.Peer.ipv4(cidr), ec2.Port.tcp(443), f"HTTPS from office {cidr}"
)
instance_sg = ec2.SecurityGroup(
self,
"InstanceSg",
vpc=vpc,
description="apm-wo grafana instance",
allow_all_outbound=True,
)
instance_sg.add_ingress_rule(
alb_sg, ec2.Port.tcp(3000), "Grafana HTTP from the ALB only"
)
# ----- Instance role: Athena query + Glue read + S3 (no static keys) -----
role = iam.Role(
self,
"GrafanaInstanceRole",
assumed_by=iam.ServicePrincipal("ec2.amazonaws.com"),
managed_policies=[
# Session Manager admin access; no SSH / bastion / key pair.
iam.ManagedPolicy.from_aws_managed_policy_name(
"AmazonSSMManagedInstanceCore"
)
],
)
role.add_to_policy(
iam.PolicyStatement(
sid="AthenaQuery",
actions=[
"athena:StartQueryExecution",
"athena:StopQueryExecution",
"athena:GetQueryExecution",
"athena:GetQueryResults",
"athena:GetWorkGroup",
"athena:ListWorkGroups",
],
resources=[
f"arn:aws:athena:{self.region}:{self.account}:workgroup/apm-wo-analysis"
],
)
)
role.add_to_policy(
iam.PolicyStatement(
sid="GlueReadOnly",
actions=[
"glue:GetDatabase",
"glue:GetDatabases",
"glue:GetTable",
"glue:GetTables",
"glue:GetPartition",
"glue:GetPartitions",
],
resources=[
f"arn:aws:glue:{self.region}:{self.account}:catalog",
f"arn:aws:glue:{self.region}:{self.account}:database/apm_wo_analysis",
f"arn:aws:glue:{self.region}:{self.account}:table/apm_wo_analysis/*",
],
)
)
bucket = s3.Bucket.from_bucket_name(self, "Exports", EXPORTS_BUCKET)
# Read the analytics snapshots; read+write athena-results (query output).
bucket.grant_read(role, "analytics/*")
bucket.grant_read_write(role, "athena-results/*")
# Config sync reads the grafana-config prefix.
bucket.grant_read(role, f"{CONFIG_PREFIX}/*")
# ----- Instance (Grafana OSS via user-data) -----
with open(USERDATA_PATH, encoding="utf-8") as fh:
userdata_script = fh.read()
userdata_script = (
userdata_script.replace("__CONFIG_BUCKET__", EXPORTS_BUCKET)
.replace("__CONFIG_PREFIX__", CONFIG_PREFIX)
.replace("__PLUGIN_VERSION__", ctx("athenaPluginVersion"))
)
user_data = ec2.UserData.for_linux()
user_data.add_commands(userdata_script)
instance = ec2.Instance(
self,
"Grafana",
vpc=vpc,
vpc_subnets=ec2.SubnetSelection(
subnet_type=ec2.SubnetType.PRIVATE_WITH_EGRESS
),
instance_type=ec2.InstanceType("t4g.small"),
machine_image=ec2.MachineImage.latest_amazon_linux2023(
cpu_type=ec2.AmazonLinuxCpuType.ARM_64
),
security_group=instance_sg,
role=role,
user_data=user_data,
require_imdsv2=True,
block_devices=[
ec2.BlockDevice(
device_name="/dev/xvda",
# RETAIN the gp3 root volume (grafana.db lives here), encrypted.
volume=ec2.BlockDeviceVolume.ebs(
20,
volume_type=ec2.EbsDeviceVolumeType.GP3,
delete_on_termination=False,
encrypted=True,
),
)
],
)
Tags.of(instance).add(BACKUP_TAG, "true")
# ----- ALB (internet-facing, office-IP-restricted, HTTPS) -----
cert = acm.Certificate.from_certificate_arn(
self, "WildcardCert", ctx("wildcardCertArn")
)
alb = elbv2.ApplicationLoadBalancer(
self,
"Alb",
vpc=vpc,
internet_facing=True,
security_group=alb_sg,
vpc_subnets=ec2.SubnetSelection(subnet_type=ec2.SubnetType.PUBLIC),
)
listener = alb.add_listener(
"Https",
port=443,
protocol=elbv2.ApplicationProtocol.HTTPS,
certificates=[cert],
# Do NOT auto-open 0.0.0.0/0 on 443 — the ALB SG already scopes
# ingress to the office CIDRs. open=True would undo that.
open=False,
)
listener.add_targets(
"GrafanaTarget",
port=3000,
protocol=elbv2.ApplicationProtocol.HTTP,
targets=[elbv2_targets.InstanceTarget(instance, 3000)],
health_check=elbv2.HealthCheck(
path="/api/health", healthy_http_codes="200"
),
)
# ----- Route53 alias grafana.seahaven.com -> ALB -----
zone = route53.HostedZone.from_hosted_zone_attributes(
self,
"SeahavenZone",
hosted_zone_id=ctx("hostedZoneId"),
zone_name=ctx("hostedZoneName"),
)
route53.ARecord(
self,
"GrafanaAlias",
zone=zone,
record_name=domain.split(".")[0],
target=route53.RecordTarget.from_alias(
route53_targets.LoadBalancerTarget(alb)
),
)
# ----- Dashboards-as-code: upload grafana/ to the S3 config prefix -----
s3deploy.BucketDeployment(
self,
"GrafanaConfig",
sources=[s3deploy.Source.asset(GRAFANA_DIR)],
destination_bucket=bucket,
destination_key_prefix=CONFIG_PREFIX,
prune=True,
)
# ----- Daily DLM snapshot of the (tagged) instance's volume -----
dlm_role = iam.Role(
self,
"DlmRole",
assumed_by=iam.ServicePrincipal("dlm.amazonaws.com"),
managed_policies=[
iam.ManagedPolicy.from_aws_managed_policy_name(
"service-role/AWSDataLifecycleManagerServiceRole"
)
],
)
dlm.CfnLifecyclePolicy(
self,
"GrafanaBackup",
description="Daily snapshot of the apm-wo grafana volume",
state="ENABLED",
execution_role_arn=dlm_role.role_arn,
policy_details=dlm.CfnLifecyclePolicy.PolicyDetailsProperty(
resource_types=["INSTANCE"],
target_tags=[CfnTag(key=BACKUP_TAG, value="true")],
schedules=[
dlm.CfnLifecyclePolicy.ScheduleProperty(
name="daily",
create_rule=dlm.CfnLifecyclePolicy.CreateRuleProperty(
interval=24, interval_unit="HOURS", times=["07:00"]
),
retain_rule=dlm.CfnLifecyclePolicy.RetainRuleProperty(count=7),
)
],
),
)