open-swe/infra/lib/constructs/app-service.ts
Adam Moussa 5fa132205b
fix(deploy): use %%...%% for CDK user-data tokens (don't collide with @@ sed) (#21)
The systemd unit booted with a literal `@@OPENSWE_ENV@@` (fetch-config.sh got the
token, not "dev" -> exit 2 -> crash-loop) because user-data.sh is double-templated:
CDK substitutes @@tokens@@ AND user-data seds @@tokens@@ into the baked
systemd/nginx files. CDK's `.replace(/@@OPENSWE_ENV@@/g, "dev")` clobbered the sed
PATTERN (`s|@@OPENSWE_ENV@@|...|` -> `s|dev|...|`, a no-op), so the unit's token
never got replaced. Same collision hit @@SERVER_NAME@@ (masked by nginx
default_server).

Fix: CDK tokens move to a DISTINCT delimiter %%...%% (rendered in app-service.ts);
the @@...@@ tokens stay for the baked-template seds. No AMI rebuild (templates
unchanged). Add a guard test asserting no unresolved %%CDK%% token survives in the
synthesized user-data. Also add .github/scripts/** to build-artifacts paths so
script-only changes trigger a publish.

jest 20/20; tsc + shellcheck clean; rendered user-data: OPENSWE_ENV="dev",
SERVER_NAME="openswe-dev.seahaven.com", @@ sed patterns preserved, 16872 B.
2026-06-26 19:06:37 -04:00

347 lines
15 KiB
TypeScript

import * as fs from "fs";
import * as path from "path";
import * as cdk from "aws-cdk-lib";
import * as ec2 from "aws-cdk-lib/aws-ec2";
import * as elbv2 from "aws-cdk-lib/aws-elasticloadbalancingv2";
import * as elbTargets from "aws-cdk-lib/aws-elasticloadbalancingv2-targets";
import * as logs from "aws-cdk-lib/aws-logs";
import * as route53 from "aws-cdk-lib/aws-route53";
import * as iam from "aws-cdk-lib/aws-iam";
import * as ssm from "aws-cdk-lib/aws-ssm";
import { Construct } from "constructs";
import { EnvName, prefix } from "../config";
import { bakedOpenSweArm64 } from "./ami-cache";
/**
* Shared seahaven-vpc + internet-facing ALB facts (read-only recon 2026-06-26;
* scratchpad/T12-infra-facts.md). A SINGLE VPC and a SINGLE shared ALB front
* both the on-prem `seahaven-site` stack and open-swe. We IMPORT every one of
* these and NEVER own them - open-swe only ADDS its own instance SG, a standalone
* ALB-egress rule, listener rules, a target group, and DNS records.
*
* ── Cross-stack coordination on the SHARED listener + ALB SG (T13 review) ──
* Two CDK stacks (seahaven-site, open-swe) add resources to the same imported
* `:443` listener and ALB SG. This is safe because each stack owns ONLY the
* resources it declares (its own logical ids): an on-prem `cdk deploy` computes a
* changeset over its own template and cannot delete rules/egress it never
* declared. The standalone-egress pattern is what on-prem itself uses
* (sgr-0c57812752a3bca13), so it does not mutate the shared SG's own definition.
*
* The ONE shared namespace that REQUIRES coordination is listener-rule PRIORITY
* (globally unique per listener; a collision is a fail-SAFE deploy error, not
* silent drift). Ownership map - keep these disjoint when editing either stack:
* - seahaven-site (on-prem): priorities 4-7 + default.
* - open-swe: priorities 2, 3 (webhooks) and 10, 11 (site).
* open-swe's webhook rules are HOST-scoped to its own *.seahaven.com hosts, so
* they never match (let alone "steal") any seahavenind.com / on-prem host.
*/
const SHARED = {
vpcId: "vpc-0d3d4b67bd0cf8a68",
availabilityZones: ["us-east-1a", "us-east-1b"],
// Private subnets host the EC2 box. The single NAT gateway lives in 1a, so the
// box is pinned to private1 (1a) for in-AZ NAT egress (no cross-AZ data $).
privateSubnetIds: ["subnet-04e38c507e96f1926", "subnet-0a0b4fc6f296dfba5"],
instanceSubnetId: "subnet-04e38c507e96f1926",
instanceAz: "us-east-1a",
albDnsName: "seahaven-com-1856441924.us-east-1.elb.amazonaws.com",
albCanonicalHostedZoneId: "Z35SXDOTRQ7X7K",
albSecurityGroupId: "sg-0b0301deed193258a",
httpsListenerArn:
"arn:aws:elasticloadbalancing:us-east-1:328440206208:listener/app/seahaven-com/222c3257354ab559/bab8bcf0da0e2927",
publicZoneId: "Z06652411XKH89KTZD3XA",
publicZoneName: "seahaven.com",
} as const;
/**
* Per-env public hostnames, listener-rule priorities, and instance size.
*
* ── Listener-rule ordering hazard (load-bearing) ──
* The shared listener already has a HOST-AGNOSTIC `/webhooks/*` PATH rule at
* priority 5 (the on-prem seahaven-site stack owns it). ALB rules are first-match
* by ASCENDING priority, so a `…/webhooks/*` request to our host would match
* rule 5 (priority 5) and be forwarded to the on-prem target BEFORE any host rule
* at 10+. Therefore our webhook rule MUST sit below priority 5. The catch-all
* "site" rule (dashboard SPA + /dashboard/api/) carries no path that collides
* with rule 5, so it can sit at any free higher number (10/11). Free priorities
* confirmed by recon: 1-3 and 8+ (4=forgejo, 5=/webhooks/*, 6/7=seahavenind).
*/
const ENV_NET: Record<
EnvName,
{
dashboardHost: string;
hooksHost: string;
webhookPriority: number;
sitePriority: number;
instanceType: string;
}
> = {
dev: {
dashboardHost: "openswe-dev.seahaven.com",
hooksHost: "hooks-dev.seahaven.com",
webhookPriority: 2,
sitePriority: 10,
instanceType: "t4g.medium",
},
prod: {
dashboardHost: "openswe.seahaven.com",
hooksHost: "hooks.seahaven.com",
webhookPriority: 3,
sitePriority: 11,
instanceType: "t4g.large",
},
};
export interface AppServiceProps {
readonly envName: EnvName;
/** Least-privilege EC2 instance role (per-env; from InstanceRole). */
readonly instanceRole: iam.IRole;
/** S3 artifact key prefix the box pulls app.tar.gz / spa.tar.gz from. */
readonly artifactPrefix?: string;
}
/**
* The open-swe compute + ingress wiring for one env (T12):
* - one ARM64 EC2 box in private1 (1a), replacement-tolerant (no RETAIN volume),
* - a standalone instance SG reachable ONLY from the shared ALB SG on :80,
* - a target group -> instance:80 (nginx is the sole ingress; :2024 stays loopback),
* - two listener rules on the imported :443 listener (webhooks below the on-prem
* path rule; site catch-all above it), both -> the same TG,
* - Route53 alias records for both hostnames -> the shared ALB,
* - IaC-owned CloudWatch log groups at 30-day retention.
*
* Everything ALB/VPC/zone-side is IMPORTED. Synth is offline: the AMI is the
* cdk.context.json-pinned AL2023 ARM64 placeholder until the baked
* open-swe-base-arm64 id is pinned before the first real deploy.
*/
export class AppService extends Construct {
public readonly instance: ec2.Instance;
public readonly targetGroup: elbv2.ApplicationTargetGroup;
/** Name of the SSM document CI fires to roll the box to the latest release. */
public readonly deployDocumentName: string;
constructor(scope: Construct, id: string, props: AppServiceProps) {
super(scope, id);
const env = props.envName;
const p = prefix(env);
const net = ENV_NET[env];
const artifactPrefix = props.artifactPrefix ?? "releases/latest";
// Import the shared VPC with explicit attributes (no fromLookup -> offline synth).
const vpc = ec2.Vpc.fromVpcAttributes(this, "Vpc", {
vpcId: SHARED.vpcId,
availabilityZones: [...SHARED.availabilityZones],
privateSubnetIds: [...SHARED.privateSubnetIds],
});
// Standalone instance SG. Egress open (NAT path); ingress only from the ALB SG.
const instanceSg = new ec2.SecurityGroup(this, "InstanceSg", {
vpc,
securityGroupName: `${p}-instance-sg`,
description: `${p} instance SG - ingress only from the shared ALB SG on :80; egress via NAT.`,
allowAllOutbound: true,
});
instanceSg.addIngressRule(
ec2.Peer.securityGroupId(SHARED.albSecurityGroupId),
ec2.Port.tcp(80),
`${p}: shared ALB SG to nginx :80`,
);
// Open the IMPORTED ALB SG to our instance via a STANDALONE egress rule, so we
// never mutate the ALB SG's own (on-prem-owned) definition.
new ec2.CfnSecurityGroupEgress(this, "AlbToInstanceEgress", {
groupId: SHARED.albSecurityGroupId,
ipProtocol: "tcp",
fromPort: 80,
toPort: 80,
destinationSecurityGroupId: instanceSg.securityGroupId,
description: `${p}: ALB to instance nginx :80`,
});
// The app-deploy procedure (deploy/ami/deploy.sh) is a normal reviewable repo
// file; CDK base64-encodes it (single line - no `$`/regex-special chars in the
// base64 alphabet) and renders it into user-data's @@DEPLOY_SH_B64@@ token, so
// user-data writes it verbatim to /opt/open-swe/bin/deploy.sh at first boot.
// The same file is run by the open-swe-<env>-deploy SSM document on every
// release - a single source of truth for "pull release, build venv, restart".
const deployShPath = path.join(__dirname, "..", "..", "..", "deploy", "ami", "deploy.sh");
// Minify before embedding: strip full-line comments + blank lines (keep the
// shebang) so the base64 fits EC2's 25.6 KB user-data limit. The repo file
// keeps its comments; only the on-box copy is minified. deploy.sh becomes
// opaque base64 here, so this never affects user-data's heredoc parsing.
const deployShMin = fs
.readFileSync(deployShPath, "utf8")
.split("\n")
.filter((line, i) => i === 0 || (!/^\s*#/.test(line) && line.trim() !== ""))
.join("\n");
const deployShB64 = Buffer.from(deployShMin, "utf8").toString("base64");
// Render the provisioning script's @@tokens@@ into the instance user-data.
// userDataCausesReplacement makes a bootstrap change roll a fresh box (the box
// holds no durable state - see ami-cache.ts / user-data.sh). Editing deploy.sh
// therefore also rolls the box (its base64 is embedded here) - acceptable: the
// box is replacement-tolerant, and ongoing releases never touch user-data.
const userDataPath = path.join(__dirname, "..", "..", "..", "deploy", "ami", "user-data.sh");
const userData = ec2.UserData.custom(
fs
.readFileSync(userDataPath, "utf8")
// %%...%% tokens are CDK-substituted here; they are DELIBERATELY a
// different delimiter from the @@...@@ tokens user-data.sh seds into the
// baked systemd/nginx templates, so CDK can never clobber a sed pattern
// (a shared @@OPENSWE_ENV@@/@@SERVER_NAME@@ left the unit unsubstituted).
.replace(/%%OPENSWE_ENV%%/g, env)
.replace(/%%ASSETS_BUCKET%%/g, `${p}-assets`)
.replace(/%%SERVER_NAME%%/g, net.dashboardHost)
.replace(/%%ARTIFACT_PREFIX%%/g, artifactPrefix)
.replace(/%%DEPLOY_SH_B64%%/g, deployShB64),
);
this.instance = new ec2.Instance(this, "Instance", {
vpc,
vpcSubnets: {
subnets: [
ec2.Subnet.fromSubnetAttributes(this, "InstanceSubnet", {
subnetId: SHARED.instanceSubnetId,
availabilityZone: SHARED.instanceAz,
}),
],
},
instanceType: new ec2.InstanceType(net.instanceType),
// The baked open-swe base AMI (deploy/ami packer build) - ARM64 Ubuntu 24.04
// with the /opt/open-swe layout, openswe user, nginx, and CW agent that
// user-data.sh assumes. Pinned by exact id (see ami-cache.ts); refresh by
// rebuilding and updating BAKED_OPEN_SWE_AMI_ID.
machineImage: bakedOpenSweArm64(),
role: props.instanceRole,
securityGroup: instanceSg,
userData,
userDataCausesReplacement: true,
requireImdsv2: true,
instanceName: `${p}-box`,
blockDevices: [
{
deviceName: "/dev/xvda",
// gp3 encrypted root; deleteOnTermination (no durable on-box state ->
// intentionally NO standalone RETAIN volume; see ami-cache.ts).
volume: ec2.BlockDeviceVolume.ebs(30, {
volumeType: ec2.EbsDeviceVolumeType.GP3,
encrypted: true,
deleteOnTermination: true,
}),
},
],
});
// SSM deploy document (open-swe-<env>-deploy): runs the baked
// /opt/open-swe/bin/deploy.sh to pull the latest release + restart. CI fires it
// (tag-scoped to project=open-swe,env=<env>) after uploading a release, so the
// app deploy role needs SendCommand ONLY on this document - NOT on the generic
// AWS-RunShellScript (closes the T4 BLOCK#3 arbitrary-shell timebox).
this.deployDocumentName = `${p}-deploy`;
new ssm.CfnDocument(this, "DeployDoc", {
name: this.deployDocumentName,
documentType: "Command",
documentFormat: "YAML",
updateMethod: "NewVersion",
content: {
schemaVersion: "2.2",
description: `Roll the ${p} box to the latest published release (runs /opt/open-swe/bin/deploy.sh).`,
mainSteps: [
{
action: "aws:runShellScript",
name: "deploy",
inputs: {
// Fixed command - no parameters, so nothing untrusted is interpolated
// into the shell. The script itself reads /etc/open-swe/boot.env.
runCommand: ["bash /opt/open-swe/bin/deploy.sh"],
},
},
],
},
});
// Target group -> instance:80 (nginx). Health check hits nginx's /healthz
// (returns 200; the dashboard TG health path defined in open-swe.nginx.conf).
this.targetGroup = new elbv2.ApplicationTargetGroup(this, "Tg", {
vpc,
targetGroupName: `${p}-tg`,
port: 80,
protocol: elbv2.ApplicationProtocol.HTTP,
targetType: elbv2.TargetType.INSTANCE,
targets: [new elbTargets.InstanceTarget(this.instance)],
deregistrationDelay: cdk.Duration.seconds(15),
healthCheck: {
path: "/healthz",
healthyHttpCodes: "200",
interval: cdk.Duration.seconds(30),
timeout: cdk.Duration.seconds(5),
healthyThresholdCount: 2,
unhealthyThresholdCount: 3,
},
});
// Import the shared :443 listener (with its ALB SG) and ADD our two rules.
const albSg = ec2.SecurityGroup.fromSecurityGroupId(this, "AlbSg", SHARED.albSecurityGroupId, {
mutable: false,
});
const listener = elbv2.ApplicationListener.fromApplicationListenerAttributes(this, "HttpsListener", {
listenerArn: SHARED.httpsListenerArn,
securityGroup: albSg,
});
// (1) Webhooks - accepted on EITHER host (integrations may target either), and
// MUST be below the on-prem path-only rule 5 (see ENV_NET note).
new elbv2.ApplicationListenerRule(this, "WebhooksRule", {
listener,
priority: net.webhookPriority,
conditions: [
elbv2.ListenerCondition.hostHeaders([net.dashboardHost, net.hooksHost]),
elbv2.ListenerCondition.pathPatterns(["/webhooks/*"]),
],
action: elbv2.ListenerAction.forward([this.targetGroup]),
});
// (2) Dashboard SPA + /dashboard/api/ (OAuth) - DASHBOARD host ONLY. The hooks
// host intentionally serves nothing but /webhooks/* (rule 1), so the OAuth /
// dashboard surface stays single-origin (OSWE-T12-02). Non-webhook paths on the
// hooks host fall through to the on-prem default.
new elbv2.ApplicationListenerRule(this, "SiteRule", {
listener,
priority: net.sitePriority,
conditions: [elbv2.ListenerCondition.hostHeaders([net.dashboardHost])],
action: elbv2.ListenerAction.forward([this.targetGroup]),
});
// Route53 ALIAS records -> the shared ALB, for both hostnames.
const zone = route53.HostedZone.fromHostedZoneAttributes(this, "PublicZone", {
hostedZoneId: SHARED.publicZoneId,
zoneName: SHARED.publicZoneName,
});
const albAlias: route53.IAliasRecordTarget = {
bind: () => ({
dnsName: SHARED.albDnsName,
hostedZoneId: SHARED.albCanonicalHostedZoneId,
}),
};
for (const [label, host] of [
["Dashboard", net.dashboardHost],
["Hooks", net.hooksHost],
] as const) {
new route53.ARecord(this, `${label}Alias`, {
zone,
recordName: host,
target: route53.RecordTarget.fromAlias(albAlias),
comment: `${p} ${label.toLowerCase()} -> shared seahaven-com ALB`,
});
}
// IaC-owned CloudWatch log groups at 30-day retention. Names mirror the
// CloudWatch-agent config (deploy/ami/templates/amazon-cloudwatch-agent.json);
// owning them here makes retention declarative rather than agent-set. Logs are
// not durable state -> DESTROY on stack delete.
for (const suffix of ["app", "user-data", "nginx-access", "nginx-error"]) {
new logs.LogGroup(this, `Log-${suffix}`, {
logGroupName: `/open-swe/${env}/${suffix}`,
retention: logs.RetentionDays.ONE_MONTH,
removalPolicy: cdk.RemovalPolicy.DESTROY,
});
}
}
}