apm-wo-analysis/tests/test_pipeline_synth.py
Adam Moussa 6ee218eea3 Add analytics dataset: projection table + Athena workgroup (Phase 3)
CDK-define the apm_wo_snapshots Glue table with partition projection over
analytics/ (dt as projected date partition, 2026-01-01..NOW). Projection means
no crawler, no MSCK REPAIR, and — critically — the classifier needs no Glue
catalog access at all.

pipeline_stack.py: glue.CfnTable (Parquet SerDe, 17-column schema mirroring the
classifier's snapshot incl. contractor_description and the two boolean flags) +
athena.CfnWorkGroup `apm-wo-analysis` (enforced result location, SSE-S3) + a
30-day lifecycle rule on athena-results/ (disposable query output in a RETAIN
bucket). Trim the classifier role: drop the entire Glue policy statement.

handler.py: stop registering the table at runtime — drop database=/table= from
to_parquet so the classifier writes pure Parquet; partition projection handles
the rest. Keeps overwrite_partitions for idempotent same-day re-uploads.

test_pipeline_synth.py: offline synth assertions (bundling skipped) — projection
properties, column schema/types, workgroup result enforcement, and that no IAM
policy grants glue:* to the classifier.

cdk synth green; 11/11 tests pass.
2026-05-28 17:24:36 -04:00

119 lines
4.3 KiB
Python

"""Synth-level assertions for the Phase 3 analytics dataset.
Synthesizes ``apm-wo-analysis-pipeline`` and asserts the Glue table carries
partition projection, the column schema matches what the classifier writes, the
Athena workgroup enforces its result location, and the classifier role has zero
Glue access (projection means it never touches the catalog). No AWS, no Docker:
``aws:cdk:bundling-stacks=[]`` skips asset bundling so this is a fast offline gate.
Run with the repo venv:
python -m pytest tests/test_pipeline_synth.py -q
"""
import sys
from pathlib import Path
import aws_cdk as cdk
from aws_cdk.assertions import Match, Template
CDK_DIR = Path(__file__).resolve().parents[1] / "cdk"
sys.path.insert(0, str(CDK_DIR))
from stacks.pipeline_stack import PipelineStack # noqa: E402
def _s3_path_ending(suffix: str):
"""Match an ``Fn::Join`` S3 path (bucket name is a Ref) ending in ``suffix``."""
return {"Fn::Join": ["", Match.array_with([suffix])]}
def _template() -> Template:
app = cdk.App(context={"aws:cdk:bundling-stacks": []})
stack = PipelineStack(
app,
"apm-wo-analysis-pipeline",
env=cdk.Environment(account="328440206208", region="us-east-1"),
)
return Template.from_stack(stack)
def test_snapshots_table_uses_partition_projection():
_template().has_resource_properties(
"AWS::Glue::Table",
{
"TableInput": {
"Name": "apm_wo_snapshots",
"PartitionKeys": [{"Name": "dt", "Type": "string"}],
"Parameters": Match.object_like(
{
"projection.enabled": "true",
"projection.dt.type": "date",
"projection.dt.format": "yyyy-MM-dd",
"projection.dt.range": "2026-01-01,NOW",
"storage.location.template": _s3_path_ending(
"/analytics/dt=${dt}/"
),
}
),
}
},
)
def test_snapshots_table_column_schema_matches_classifier():
# Booleans typed correctly and the columns the BUILD.md prose omitted
# (contractor_description) are present — schema mirrors the writer.
_template().has_resource_properties(
"AWS::Glue::Table",
{
"TableInput": {
"StorageDescriptor": Match.object_like(
{
# array_with is order-sensitive: list patterns in the
# same order the classifier writes them.
"Columns": Match.array_with(
[
{"Name": "contractor_description", "Type": "string"},
{"Name": "is_escalation", "Type": "boolean"},
{"Name": "is_action", "Type": "boolean"},
{"Name": "mismatch", "Type": "string"},
]
),
}
),
}
},
)
def test_athena_workgroup_enforces_results_location():
_template().has_resource_properties(
"AWS::Athena::WorkGroup",
{
"Name": "apm-wo-analysis",
"WorkGroupConfiguration": Match.object_like(
{
"EnforceWorkGroupConfiguration": True,
"ResultConfiguration": {
"OutputLocation": _s3_path_ending("/athena-results/"),
"EncryptionConfiguration": {"EncryptionOption": "SSE_S3"},
},
}
),
},
)
def test_classifier_role_has_no_glue_access():
# Partition projection => the classifier only writes Parquet to S3. No IAM
# policy in the stack should grant any glue:* action.
policies = _template().find_resources("AWS::IAM::Policy")
for policy in policies.values():
for stmt in policy["Properties"]["PolicyDocument"]["Statement"]:
actions = stmt.get("Action", [])
actions = [actions] if isinstance(actions, str) else actions
offending = [
a for a in actions if isinstance(a, str) and a.startswith("glue:")
]
assert not offending, f"unexpected Glue access: {offending}"