import * as ec2 from "aws-cdk-lib/aws-ec2"; /** * cdk.context.json key for the cached AL2023 ARM64 AMI. `latestAmazonLinux2023` * + ARM_64 in the pinned aws-cdk-lib resolves the public SSM parameter below * (the default-kernel alias for this CDK version is the `kernel-6.1` line); with * `cachedInContext: true` CDK stores the resolved id under this exact key. * Exported so the env stack can check "is the AMI already pinned?" and skip the * resolve at synth when it is not — guaranteeing T3 synth makes no live AWS call. * * If an aws-cdk-lib bump changes the default kernel alias, `cdk synth` will write * a new key into cdk.context.json — update this constant + the committed pin to * match (the AMI cache naturally tracks the CDK version). */ export const AL2023_ARM64_SSM_CONTEXT_KEY = "ssm:account=328440206208:parameterName=/aws/service/ami-amazon-linux-latest/al2023-ami-kernel-6.1-arm64:region=us-east-1"; /** * Cached ARM64 Amazon Linux 2023 machine image. * * ── EBS / AMI cache discipline (see memory feedback_inline_ebs_volumes) ── * * `cachedInContext: true` PINS the resolved AMI id into the committed * cdk.context.json. Without it, `latestAmazonLinux2023()` resolves the NEWEST * AL2023 release on every synth/deploy, so a routine deploy can swap the AMI → * EC2 instance REPLACEMENT whenever AWS ships a release. That was the root cause * of the file-share data-loss incidents (5/15, 5/27, 6/5). Refresh the pin * DELIBERATELY: * * cdk context --reset '' && cdk synth * * then review the `cdk diff` (it WILL report "requires replacement") before * deploying. * * ── userDataCausesReplacement intent (consumed at T12) ── * * open-swe user-data is provisioning-only: install the langgraph runtime, pull * config from Secrets Manager / SSM, pull the build artifact from S3, start the * service. It holds NO durable state. T12 sets `userDataCausesReplacement: true` * DELIBERATELY so a config/bootstrap change rolls a fresh, known-good box. * * ── "No durable state on box → no RETAIN volume" assertion ── * * The langgraph store is in-memory and is reconstructed on every boot from S3 * (artifact) + Secrets Manager / SSM (config). Nothing of record lives only on * the instance's disk. Therefore there is intentionally NO standalone * `ec2.Volume` + `removalPolicy.RETAIN` here: the design goal is replacement- * TOLERANCE, not replacement-avoidance. * * ── Operational guard still applies at T12 (feedback_inline_ebs_volumes) ── * * Before ANY replacing deploy (AMI / userData / instance-type change): snapshot * the root volume AND wait for `state=completed`, re-verify "no local-only * durable state" first, and ensure cdk-diff-on-PR surfaces the replacement at * review time. */ export function cachedArm64AmazonLinux2023(): ec2.IMachineImage { return ec2.MachineImage.latestAmazonLinux2023({ cpuType: ec2.AmazonLinuxCpuType.ARM_64, cachedInContext: true, }); }