mirror of
https://github.com/Sea-Haven-Industries/apm-wo-analysis.git
synced 2026-09-30 07:43:15 +00:00
CDK-define the apm_wo_snapshots Glue table with partition projection over analytics/ (dt as projected date partition, 2026-01-01..NOW). Projection means no crawler, no MSCK REPAIR, and — critically — the classifier needs no Glue catalog access at all. pipeline_stack.py: glue.CfnTable (Parquet SerDe, 17-column schema mirroring the classifier's snapshot incl. contractor_description and the two boolean flags) + athena.CfnWorkGroup `apm-wo-analysis` (enforced result location, SSE-S3) + a 30-day lifecycle rule on athena-results/ (disposable query output in a RETAIN bucket). Trim the classifier role: drop the entire Glue policy statement. handler.py: stop registering the table at runtime — drop database=/table= from to_parquet so the classifier writes pure Parquet; partition projection handles the rest. Keeps overwrite_partitions for idempotent same-day re-uploads. test_pipeline_synth.py: offline synth assertions (bundling skipped) — projection properties, column schema/types, workgroup result enforcement, and that no IAM policy grants glue:* to the classifier. cdk synth green; 11/11 tests pass.
119 lines
4.3 KiB
Python
119 lines
4.3 KiB
Python
"""Synth-level assertions for the Phase 3 analytics dataset.
|
|
|
|
Synthesizes ``apm-wo-analysis-pipeline`` and asserts the Glue table carries
|
|
partition projection, the column schema matches what the classifier writes, the
|
|
Athena workgroup enforces its result location, and the classifier role has zero
|
|
Glue access (projection means it never touches the catalog). No AWS, no Docker:
|
|
``aws:cdk:bundling-stacks=[]`` skips asset bundling so this is a fast offline gate.
|
|
|
|
Run with the repo venv:
|
|
|
|
python -m pytest tests/test_pipeline_synth.py -q
|
|
"""
|
|
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import aws_cdk as cdk
|
|
from aws_cdk.assertions import Match, Template
|
|
|
|
CDK_DIR = Path(__file__).resolve().parents[1] / "cdk"
|
|
sys.path.insert(0, str(CDK_DIR))
|
|
|
|
from stacks.pipeline_stack import PipelineStack # noqa: E402
|
|
|
|
|
|
def _s3_path_ending(suffix: str):
|
|
"""Match an ``Fn::Join`` S3 path (bucket name is a Ref) ending in ``suffix``."""
|
|
return {"Fn::Join": ["", Match.array_with([suffix])]}
|
|
|
|
|
|
def _template() -> Template:
|
|
app = cdk.App(context={"aws:cdk:bundling-stacks": []})
|
|
stack = PipelineStack(
|
|
app,
|
|
"apm-wo-analysis-pipeline",
|
|
env=cdk.Environment(account="328440206208", region="us-east-1"),
|
|
)
|
|
return Template.from_stack(stack)
|
|
|
|
|
|
def test_snapshots_table_uses_partition_projection():
|
|
_template().has_resource_properties(
|
|
"AWS::Glue::Table",
|
|
{
|
|
"TableInput": {
|
|
"Name": "apm_wo_snapshots",
|
|
"PartitionKeys": [{"Name": "dt", "Type": "string"}],
|
|
"Parameters": Match.object_like(
|
|
{
|
|
"projection.enabled": "true",
|
|
"projection.dt.type": "date",
|
|
"projection.dt.format": "yyyy-MM-dd",
|
|
"projection.dt.range": "2026-01-01,NOW",
|
|
"storage.location.template": _s3_path_ending(
|
|
"/analytics/dt=${dt}/"
|
|
),
|
|
}
|
|
),
|
|
}
|
|
},
|
|
)
|
|
|
|
|
|
def test_snapshots_table_column_schema_matches_classifier():
|
|
# Booleans typed correctly and the columns the BUILD.md prose omitted
|
|
# (contractor_description) are present — schema mirrors the writer.
|
|
_template().has_resource_properties(
|
|
"AWS::Glue::Table",
|
|
{
|
|
"TableInput": {
|
|
"StorageDescriptor": Match.object_like(
|
|
{
|
|
# array_with is order-sensitive: list patterns in the
|
|
# same order the classifier writes them.
|
|
"Columns": Match.array_with(
|
|
[
|
|
{"Name": "contractor_description", "Type": "string"},
|
|
{"Name": "is_escalation", "Type": "boolean"},
|
|
{"Name": "is_action", "Type": "boolean"},
|
|
{"Name": "mismatch", "Type": "string"},
|
|
]
|
|
),
|
|
}
|
|
),
|
|
}
|
|
},
|
|
)
|
|
|
|
|
|
def test_athena_workgroup_enforces_results_location():
|
|
_template().has_resource_properties(
|
|
"AWS::Athena::WorkGroup",
|
|
{
|
|
"Name": "apm-wo-analysis",
|
|
"WorkGroupConfiguration": Match.object_like(
|
|
{
|
|
"EnforceWorkGroupConfiguration": True,
|
|
"ResultConfiguration": {
|
|
"OutputLocation": _s3_path_ending("/athena-results/"),
|
|
"EncryptionConfiguration": {"EncryptionOption": "SSE_S3"},
|
|
},
|
|
}
|
|
),
|
|
},
|
|
)
|
|
|
|
|
|
def test_classifier_role_has_no_glue_access():
|
|
# Partition projection => the classifier only writes Parquet to S3. No IAM
|
|
# policy in the stack should grant any glue:* action.
|
|
policies = _template().find_resources("AWS::IAM::Policy")
|
|
for policy in policies.values():
|
|
for stmt in policy["Properties"]["PolicyDocument"]["Statement"]:
|
|
actions = stmt.get("Action", [])
|
|
actions = [actions] if isinstance(actions, str) else actions
|
|
offending = [
|
|
a for a in actions if isinstance(a, str) and a.startswith("glue:")
|
|
]
|
|
assert not offending, f"unexpected Glue access: {offending}"
|