open-swe/infra/lib/constructs/instance-role.ts

161 lines
7.6 KiB
TypeScript
Raw Normal View History

import * as iam from "aws-cdk-lib/aws-iam";
import { Construct } from "constructs";
import { ACCOUNT, EnvName, REGION, prefix } from "../config";
/**
* Least-privilege EC2 instance role for the open-swe box (one per env).
*
* Grants exactly what the boot/runtime flow needs and NOTHING ELSE — no admin,
* no `*` resources except where the AWS action genuinely has no resource-level
* scoping. Per-env so the dev box can never read prod secrets/config and vice
* versa. Reviewed at T4 (GPT-4.1 IAM cross-review) / T5 (/sh-security-review)
* before it is ever deployed (T6).
*/
export class InstanceRole extends Construct {
public readonly role: iam.Role;
constructor(scope: Construct, id: string, env: EnvName) {
super(scope, id);
const p = prefix(env);
this.role = new iam.Role(this, "Role", {
roleName: `${p}-instance-role`,
assumedBy: new iam.ServicePrincipal("ec2.amazonaws.com"),
description: `EC2 instance role for the ${p} open-swe box (least-privilege).`,
});
// AWS-managed: lets the SSM agent register the instance and RECEIVE the
// app-deploy `ssm:SendCommand` from githubdeploy-open-swe-app. This is the
// standard Session-Manager / RunCommand grant and is the only managed
// policy on the role. DELIBERATE — flag for T4 confirmation.
this.role.addManagedPolicy(
iam.ManagedPolicy.fromAwsManagedPolicyName("AmazonSSMManagedInstanceCore"),
);
// Read the build artifact from the env's S3 asset bucket (deploy = pull).
feat: stand up dev properly — assets bucket + artifact CD + baked AMI + on-box uv sync (T7+T19+T14) (#18) * feat(infra): build + pin the baked open-swe-base-arm64 AMI (T12 AMI / item 3) Packer-build the custom base image and repoint AppService off the AL2023 placeholder onto it. deploy/ami/open-swe-base.pkr.hcl — fix two bugs that blocked the first real `packer build` (the config had only ever been `packer validate`'d at T8): - the file provisioner failed uploading the templates dir ('scp: …: Is a directory') — a trailing-slash contents-upload needs the dest dir to exist; added a 'mkdir -p /tmp/open-swe-templates' shell provisioner + dropped the dest trailing slash. - the shell provisioner's custom execute_command omitted {{ .Vars }}, so the environment_vars never reached provision.sh (which runs under set -u and aborted on CLOUDWATCH_AGENT_DEB_URL). Added {{ .Vars }}. infra: - ami-cache.ts: BAKED_OPEN_SWE_AMI_ID = ami-0545363bb147229ff (built 2026-06-26 from open-swe-base-arm64-20260626-201929) + bakedOpenSweArm64() pinning it by exact id via MachineImage.genericLinux (offline, deterministic). Dropped the now-dead AL2023 cachedInContext helper + context key; kept the EBS/replacement discipline docs. - app-service.ts: machineImage → bakedOpenSweArm64(). - open-swe-stack.ts: output BakedAmiId (was the AL2023 PinnedAmiId guard). - cdk.context.json → {} (AMI is a static id pin; no context lookups remain). - README: Baked AMI + EBS-replacement-discipline section. tsc + cdk synth(dev+prod) + jest(16) clean; template ImageId = the baked AMI. NOTE: held — do NOT merge until the open-swe-dev secret values are populated (put-config.sh). The infra CD is live, so merging this to dev auto-deploys OpenSweDevStack; without secrets the box boots but fetch-config fail-fasts → unhealthy ALB target on the shared prod ALB. Merge once secrets are set (T14). * fix(ami): ASCII-only AMI description + re-pin to ami-00080084502093021 Third packer bug: ami_description had an em-dash (non-ASCII); AWS rejects non-ASCII in the AMI Description attribute, so packer registered then DEREGISTERED the first AMI (ami-0545…) on the ModifyImageAttribute error. Replaced with an ASCII '-'. Rebuilt clean → ami-00080084502093021 (available). Re-pinned BAKED_OPEN_SWE_AMI_ID. * fix(deploy): GitHub App + Slack required for prod only, not dev Per the migration decision: do NOT create/duplicate a separate dev GitHub App or Slack app — only prod owns the single shared app. So fetch-config.sh no longer hard-requires the GitHub App quintet (ID/PRIVATE_KEY/INSTALLATION_ID/CLIENT_ID/ CLIENT_SECRET) + Slack/webhook secrets for dev; they move into the prod-only block alongside the existing GITHUB_WEBHOOK_SECRET/SLACK_SIGNING_SECRET. Dev now boots with just DASHBOARD_JWT_SECRET + TOKEN_ENCRYPTION_KEY + the active provider key(s) + the langsmith sandbox keys. Dev is a deployment-validation env (boot/health/boundary) with no GitHub/Slack/webhook integration; prod parity is unchanged (prod still requires everything). * feat: stand up dev properly — S3 assets bucket + artifact CD + on-box uv sync (T7+T19) Make the dev/prod box deployable end-to-end: a real artifact pipeline and a re-runnable on-box deploy, so OpenSweDevStack can come up genuinely healthy. Infra (T7): - assets-bucket.ts: open-swe-<env>-assets S3 bucket — BLOCK_ALL public access, SSE-S3, enforceSSL (deny non-TLS), versioned, lifecycle (expire noncurrent + abort MPU), RETAIN. Wired into OpenSweStack + CfnOutput. - app-service.ts: open-swe-<env>-deploy SSM document that runs the baked /opt/open-swe/bin/deploy.sh (tag-scoped roll-the-box). machineImage is the baked open-swe-base-arm64 AMI (folds in the held #16). IAM (app deploy role — cross-review gated): - github-deploy-roles.ts: app role gains s3:PutObject/DeleteObject scoped to open-swe-<env>-assets/releases/* (CI uploads releases). Drops the generic AWS-RunShellScript grant now that the dedicated open-swe-<env>-deploy document is the only SendCommand path — closes the T4 BLOCK#3 arbitrary-shell timebox. Boot/deploy (T19): - deploy/ami/deploy.sh: single, re-runnable app-deploy procedure — pull app.tar.gz/spa.tar.gz from S3, `uv sync --frozen --no-dev` (native ARM64 venv at the real path, py3.12 pre-baked), restart open-swe.service + reload nginx. - user-data.sh: nginx starts BEFORE the app deploy (static /healthz -> the ALB target is healthy even before the first release); deploy.sh is base64-rendered by CDK into user-data (a normal reviewable repo file, not a heredoc) and the first-boot deploy is NON-FATAL (no release yet -> wait for the first SSM deploy). CI (T7+T19): - build-artifacts.yml (+ .github/scripts): build the SPA with bun (vite -> ui/.output/public -> spa.tar.gz), package the Python source via git archive (app.tar.gz, no ui/ no .venv), upload to releases/<sha>/ + releases/latest/ via the githubdeploy-open-swe-app-<env> OIDC role, then fire open-swe-<env>-deploy. push dev -> dev (auto); push main -> prod (env "prod" approval gate). Local: ruff/shellcheck clean, tsc clean, jest 16/16, cdk synth offline OK, deploy.sh base64 round-trips exact. * harden(sec-review): tar extraction, deploy gating, least-privilege, secret guard Address the /sh-security-review fan-out + proof-or-kill verifier pass. Only one confirmed-high surfaced and it is PRE-EXISTING and out-of-diff (OSWE-IAC-AUDIT-01, the account-wide CDK cfn-exec residual already documented in config.ts; recorded in .security-review/suppressions.json with justification + flagged for the per-env bootstrap-qualifier follow-up). The rest were verifier-downgraded to unverified; these are the cheap defense-in-depth fixes worth taking regardless: - deploy.sh: extract tarballs with --no-same-owner --no-same-permissions (root never honors an archive's uid/mode → no setuid/foreign-owned file can land); and treat "no release in S3 yet" as a benign exit 0, distinct from a real deploy failure (set -e stays loud once a release exists). - publish-and-deploy.sh: gate on the AGGREGATE SSM Command.Status (+ TargetCount), not CommandInvocations[0], so a partial failure across the brief 2-instance replacement window can't be reported as success. - instance-role.ts: scope the box's s3:GetObject to releases/* (mirrors the app role's write scope) instead of the whole bucket. - package-artifacts.sh: fail-closed secret-shaped-file guard on app.tar.gz (defense in depth over .gitignore; scoped to data extensions so *_credentials.py source is not a false positive — verified against the real tree). Deferred as documented follow-ups (verifier: unverified, supply-chain-gated to the CI OIDC writer; bucket is BLOCK_ALL + enforceSSL + versioned): SHA-pinned immutable releases/<sha>/ pulls + signed checksum (vs mutable latest/), single-tarball release to remove the torn-read window, and app-aware ALB health (vs static nginx /healthz). shellcheck/tsc/jest(16) clean; both stacks synth offline. * fix(infra): ASCII-only EC2 SecurityGroup descriptions + synth-time guard The instance-SG GroupDescription + ingress/egress rule descriptions carried an em-dash / arrow (—, →). `tsc` and `cdk synth` accept them, but the EC2 API rejects non-ASCII in GroupDescription ("Character sets beyond ASCII are not supported"), so OpenSweDevStack's first deploy failed at the SG and rolled back. (Pre-existing from #14; same class as the AMI-description ASCII bug.) - app-service.ts: replace —/→ with ASCII (- / ->) in the SG GroupDescription, the ingress/egress rule descriptions, and the Route53 comment. - test/ascii-aws-fields.test.ts: synth-time guard asserting EC2 SecurityGroup GroupDescription + rule descriptions are pure ASCII, so this fails the build instead of a deploy next time. jest 18/18; tsc clean. * fix(infra): SG rule descriptions use ASCII-charset-safe text (no `>`) The first ASCII fix replaced the arrow with `->`, but EC2 SecurityGroup *rule* descriptions allow a stricter set than ASCII — `a-zA-Z0-9. _-:/()#,@[]+=&;{}!$*`, which EXCLUDES `<`/`>`. So OpenSweDevStack's second deploy still failed at the ingress rule. Use "to" instead of "->", and tighten the guard test from "ASCII only" to the exact EC2 allowed charset so it catches `>` (and `<`) too. jest 18/18; tsc clean. * fix(infra): minify embedded deploy.sh so user-data fits EC2's 25.6 KB limit The base64 deploy.sh embedded in user-data pushed the encoded boot script to 27184 bytes, over EC2's 25600-byte cap, so OpenSweDevStack's instance failed with "Encoded User data is limited to 25600 bytes". Strip full-line comments + blank lines from deploy.sh before base64-embedding it (repo file keeps comments; only the on-box copy is minified; the script is opaque base64 so user-data heredocs are unaffected) -> rendered user-data drops to 16424 bytes (9 KB margin). Add a synth-time guard test asserting EC2 user-data stays under 25600 bytes encoded. jest 19/19; minified deploy.sh passes bash -n + shellcheck.
2026-06-26 18:49:09 -04:00
// Scoped to releases/* — the only prefix CI writes and the box pulls — so a
// compromised box (or stolen IMDS creds) cannot read anything else that might
// ever land in the bucket (least-privilege; mirrors the app role's write scope).
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "ReadArtifactObjects",
actions: ["s3:GetObject"],
feat: stand up dev properly — assets bucket + artifact CD + baked AMI + on-box uv sync (T7+T19+T14) (#18) * feat(infra): build + pin the baked open-swe-base-arm64 AMI (T12 AMI / item 3) Packer-build the custom base image and repoint AppService off the AL2023 placeholder onto it. deploy/ami/open-swe-base.pkr.hcl — fix two bugs that blocked the first real `packer build` (the config had only ever been `packer validate`'d at T8): - the file provisioner failed uploading the templates dir ('scp: …: Is a directory') — a trailing-slash contents-upload needs the dest dir to exist; added a 'mkdir -p /tmp/open-swe-templates' shell provisioner + dropped the dest trailing slash. - the shell provisioner's custom execute_command omitted {{ .Vars }}, so the environment_vars never reached provision.sh (which runs under set -u and aborted on CLOUDWATCH_AGENT_DEB_URL). Added {{ .Vars }}. infra: - ami-cache.ts: BAKED_OPEN_SWE_AMI_ID = ami-0545363bb147229ff (built 2026-06-26 from open-swe-base-arm64-20260626-201929) + bakedOpenSweArm64() pinning it by exact id via MachineImage.genericLinux (offline, deterministic). Dropped the now-dead AL2023 cachedInContext helper + context key; kept the EBS/replacement discipline docs. - app-service.ts: machineImage → bakedOpenSweArm64(). - open-swe-stack.ts: output BakedAmiId (was the AL2023 PinnedAmiId guard). - cdk.context.json → {} (AMI is a static id pin; no context lookups remain). - README: Baked AMI + EBS-replacement-discipline section. tsc + cdk synth(dev+prod) + jest(16) clean; template ImageId = the baked AMI. NOTE: held — do NOT merge until the open-swe-dev secret values are populated (put-config.sh). The infra CD is live, so merging this to dev auto-deploys OpenSweDevStack; without secrets the box boots but fetch-config fail-fasts → unhealthy ALB target on the shared prod ALB. Merge once secrets are set (T14). * fix(ami): ASCII-only AMI description + re-pin to ami-00080084502093021 Third packer bug: ami_description had an em-dash (non-ASCII); AWS rejects non-ASCII in the AMI Description attribute, so packer registered then DEREGISTERED the first AMI (ami-0545…) on the ModifyImageAttribute error. Replaced with an ASCII '-'. Rebuilt clean → ami-00080084502093021 (available). Re-pinned BAKED_OPEN_SWE_AMI_ID. * fix(deploy): GitHub App + Slack required for prod only, not dev Per the migration decision: do NOT create/duplicate a separate dev GitHub App or Slack app — only prod owns the single shared app. So fetch-config.sh no longer hard-requires the GitHub App quintet (ID/PRIVATE_KEY/INSTALLATION_ID/CLIENT_ID/ CLIENT_SECRET) + Slack/webhook secrets for dev; they move into the prod-only block alongside the existing GITHUB_WEBHOOK_SECRET/SLACK_SIGNING_SECRET. Dev now boots with just DASHBOARD_JWT_SECRET + TOKEN_ENCRYPTION_KEY + the active provider key(s) + the langsmith sandbox keys. Dev is a deployment-validation env (boot/health/boundary) with no GitHub/Slack/webhook integration; prod parity is unchanged (prod still requires everything). * feat: stand up dev properly — S3 assets bucket + artifact CD + on-box uv sync (T7+T19) Make the dev/prod box deployable end-to-end: a real artifact pipeline and a re-runnable on-box deploy, so OpenSweDevStack can come up genuinely healthy. Infra (T7): - assets-bucket.ts: open-swe-<env>-assets S3 bucket — BLOCK_ALL public access, SSE-S3, enforceSSL (deny non-TLS), versioned, lifecycle (expire noncurrent + abort MPU), RETAIN. Wired into OpenSweStack + CfnOutput. - app-service.ts: open-swe-<env>-deploy SSM document that runs the baked /opt/open-swe/bin/deploy.sh (tag-scoped roll-the-box). machineImage is the baked open-swe-base-arm64 AMI (folds in the held #16). IAM (app deploy role — cross-review gated): - github-deploy-roles.ts: app role gains s3:PutObject/DeleteObject scoped to open-swe-<env>-assets/releases/* (CI uploads releases). Drops the generic AWS-RunShellScript grant now that the dedicated open-swe-<env>-deploy document is the only SendCommand path — closes the T4 BLOCK#3 arbitrary-shell timebox. Boot/deploy (T19): - deploy/ami/deploy.sh: single, re-runnable app-deploy procedure — pull app.tar.gz/spa.tar.gz from S3, `uv sync --frozen --no-dev` (native ARM64 venv at the real path, py3.12 pre-baked), restart open-swe.service + reload nginx. - user-data.sh: nginx starts BEFORE the app deploy (static /healthz -> the ALB target is healthy even before the first release); deploy.sh is base64-rendered by CDK into user-data (a normal reviewable repo file, not a heredoc) and the first-boot deploy is NON-FATAL (no release yet -> wait for the first SSM deploy). CI (T7+T19): - build-artifacts.yml (+ .github/scripts): build the SPA with bun (vite -> ui/.output/public -> spa.tar.gz), package the Python source via git archive (app.tar.gz, no ui/ no .venv), upload to releases/<sha>/ + releases/latest/ via the githubdeploy-open-swe-app-<env> OIDC role, then fire open-swe-<env>-deploy. push dev -> dev (auto); push main -> prod (env "prod" approval gate). Local: ruff/shellcheck clean, tsc clean, jest 16/16, cdk synth offline OK, deploy.sh base64 round-trips exact. * harden(sec-review): tar extraction, deploy gating, least-privilege, secret guard Address the /sh-security-review fan-out + proof-or-kill verifier pass. Only one confirmed-high surfaced and it is PRE-EXISTING and out-of-diff (OSWE-IAC-AUDIT-01, the account-wide CDK cfn-exec residual already documented in config.ts; recorded in .security-review/suppressions.json with justification + flagged for the per-env bootstrap-qualifier follow-up). The rest were verifier-downgraded to unverified; these are the cheap defense-in-depth fixes worth taking regardless: - deploy.sh: extract tarballs with --no-same-owner --no-same-permissions (root never honors an archive's uid/mode → no setuid/foreign-owned file can land); and treat "no release in S3 yet" as a benign exit 0, distinct from a real deploy failure (set -e stays loud once a release exists). - publish-and-deploy.sh: gate on the AGGREGATE SSM Command.Status (+ TargetCount), not CommandInvocations[0], so a partial failure across the brief 2-instance replacement window can't be reported as success. - instance-role.ts: scope the box's s3:GetObject to releases/* (mirrors the app role's write scope) instead of the whole bucket. - package-artifacts.sh: fail-closed secret-shaped-file guard on app.tar.gz (defense in depth over .gitignore; scoped to data extensions so *_credentials.py source is not a false positive — verified against the real tree). Deferred as documented follow-ups (verifier: unverified, supply-chain-gated to the CI OIDC writer; bucket is BLOCK_ALL + enforceSSL + versioned): SHA-pinned immutable releases/<sha>/ pulls + signed checksum (vs mutable latest/), single-tarball release to remove the torn-read window, and app-aware ALB health (vs static nginx /healthz). shellcheck/tsc/jest(16) clean; both stacks synth offline. * fix(infra): ASCII-only EC2 SecurityGroup descriptions + synth-time guard The instance-SG GroupDescription + ingress/egress rule descriptions carried an em-dash / arrow (—, →). `tsc` and `cdk synth` accept them, but the EC2 API rejects non-ASCII in GroupDescription ("Character sets beyond ASCII are not supported"), so OpenSweDevStack's first deploy failed at the SG and rolled back. (Pre-existing from #14; same class as the AMI-description ASCII bug.) - app-service.ts: replace —/→ with ASCII (- / ->) in the SG GroupDescription, the ingress/egress rule descriptions, and the Route53 comment. - test/ascii-aws-fields.test.ts: synth-time guard asserting EC2 SecurityGroup GroupDescription + rule descriptions are pure ASCII, so this fails the build instead of a deploy next time. jest 18/18; tsc clean. * fix(infra): SG rule descriptions use ASCII-charset-safe text (no `>`) The first ASCII fix replaced the arrow with `->`, but EC2 SecurityGroup *rule* descriptions allow a stricter set than ASCII — `a-zA-Z0-9. _-:/()#,@[]+=&;{}!$*`, which EXCLUDES `<`/`>`. So OpenSweDevStack's second deploy still failed at the ingress rule. Use "to" instead of "->", and tighten the guard test from "ASCII only" to the exact EC2 allowed charset so it catches `>` (and `<`) too. jest 18/18; tsc clean. * fix(infra): minify embedded deploy.sh so user-data fits EC2's 25.6 KB limit The base64 deploy.sh embedded in user-data pushed the encoded boot script to 27184 bytes, over EC2's 25600-byte cap, so OpenSweDevStack's instance failed with "Encoded User data is limited to 25600 bytes". Strip full-line comments + blank lines from deploy.sh before base64-embedding it (repo file keeps comments; only the on-box copy is minified; the script is opaque base64 so user-data heredocs are unaffected) -> rendered user-data drops to 16424 bytes (9 KB margin). Add a synth-time guard test asserting EC2 user-data stays under 25600 bytes encoded. jest 19/19; minified deploy.sh passes bash -n + shellcheck.
2026-06-26 18:49:09 -04:00
resources: [`arn:aws:s3:::${p}-assets/releases/*`],
}),
);
fix: resolve security-review findings (sandbox isolation, IAM list scope, webhook replay, info-leak) (#54) * fix: enforce a replay window on Linear webhooks (AUTHZ-001) verify_linear_signature accepted any correctly-signed body with no freshness check, so a captured request could be replayed indefinitely. Parse the signed webhookTimestamp (Unix ms) and reject requests outside a 60s window, failing closed when the field is missing or malformed — mirroring the Slack verifier. * fix: stop leaking upstream auth-error bodies into user comments get_github_token_for_user folded the raw upstream response text into the error string that becomes a Slack/Linear comment (AUTH-RESP-LEAK-01). Log the full body server-side only and return a generic "GitHub auth failed (status <code>)". Also document the accepted shared-installation-token blast radius on the bot-token-only path (AUTHZ-003). * fix: bind sandbox and token caches to repo to prevent thread-id collision A PR head-branch name is attacker-controllable and get_thread_id_from_branch derives a thread_id from its first UUID with no repo binding (TID-COLLIDE-01). The in-memory sandbox cache and the per-thread GitHub-token cache were keyed on thread_id alone, and a cached sandbox was reused after only an echo-ping, so a different repo's webhook could bind to another thread's sandbox or token. Without changing the persistent thread-id scheme: - Persist the bound repo (owner/name) in thread metadata on sandbox creation and refuse to reuse a sandbox whose bound repo does not match the current event (SandboxRepoMismatchError); the in-memory proxy also carries the binding. - Bind the GitHub-token cache entries to their repo and evict on a cross-repo read so a colliding thread_id cannot be served another repo's token. - Thread repo through the reviewer and the webhook token resolvers. * fix: scope s3:ListBucket to the releases/ prefix (F-1/IAC-04) The instance role and the GitHub deploy app role granted s3:ListBucket on the whole assets bucket. Every caller (deploy.sh, the publish/rollback scripts) only ever lists under releases/, so add a StringLike s3:prefix=releases/* condition. GetBucketLocation has no s3:prefix in its request context, so it moves to its own unconditioned statement. Also document the accepted F-2 cross-env existence-oracle residual on BatchGetSecretValue. * chore: suppress test-fixture credential false positive; document AUTHZ-002 Add a machine-level suppression for the fake Datadog key in the test_team_credentials encryption-roundtrip fixture (CWE-798, not a real credential). Clarify that the within-org thread-write path is intentional by design (AUTHZ-002) — comment only, no behavior change. * fix: casefold repo-binding keys to avoid spurious cross-repo mismatch GitHub owner/name are case-insensitive. Casefold the owner/name key on both the write (binding) and read (compare) sides — repo_cache_key and the metadata bound_repo read — so Org/Repo and org/repo resolve to one repo and a legitimate same-repo run cannot raise a spurious SandboxRepoMismatchError (Gap 2). * fix: stop leaking upstream auth body in unexpected-result branch The 2xx-but-missing-token/url branch echoed the parsed upstream response body into the user-facing error. Return a generic message and log response_data server-side only, mirroring the existing HTTPStatusError fix (Gap 4). * fix: fail closed for unbound-legacy sandboxes and catch repo mismatch Gap 1: a thread with a persisted sandbox_id but no in-memory cache and no recorded bound_repo (a pre-binding legacy thread, post-deploy) previously reconnected-and-served the sandbox to the current repo, then rebound it. Now fail closed: drop the stale id and recreate a fresh sandbox bound to this repo, logging a reconnect-with-missing-binding event. A sandbox is never served to a repo unless its binding is known and matches; new threads bind on first run unchanged. Gap 3: catch SandboxRepoMismatchError at the agent and reviewer run entrypoints, log it for alarming, and surface a clean sanitized error instead of letting an opaque deep-stack exception crash-loop the worker. * chore: suppress test-fixture credential false positive in token-TTL tests Add a machine-level suppression for the fake "ghp_secret" GitHub token used by the cached-token TTL/revocation unit tests (CWE-798). Not a real credential and not a valid PAT; scoped to the unit test only.
2026-06-29 12:21:19 -04:00
// ListBucket is constrained to the releases/ prefix (F-1/IAC-04) — the box
// only ever lists release artifacts, so a compromised box cannot enumerate
// any other object that might land in the bucket. GetBucketLocation has no
// s3:prefix in its request context, so it stays a separate, unconditioned
// statement (the condition would otherwise AccessDeny it).
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "ListArtifactBucket",
fix: resolve security-review findings (sandbox isolation, IAM list scope, webhook replay, info-leak) (#54) * fix: enforce a replay window on Linear webhooks (AUTHZ-001) verify_linear_signature accepted any correctly-signed body with no freshness check, so a captured request could be replayed indefinitely. Parse the signed webhookTimestamp (Unix ms) and reject requests outside a 60s window, failing closed when the field is missing or malformed — mirroring the Slack verifier. * fix: stop leaking upstream auth-error bodies into user comments get_github_token_for_user folded the raw upstream response text into the error string that becomes a Slack/Linear comment (AUTH-RESP-LEAK-01). Log the full body server-side only and return a generic "GitHub auth failed (status <code>)". Also document the accepted shared-installation-token blast radius on the bot-token-only path (AUTHZ-003). * fix: bind sandbox and token caches to repo to prevent thread-id collision A PR head-branch name is attacker-controllable and get_thread_id_from_branch derives a thread_id from its first UUID with no repo binding (TID-COLLIDE-01). The in-memory sandbox cache and the per-thread GitHub-token cache were keyed on thread_id alone, and a cached sandbox was reused after only an echo-ping, so a different repo's webhook could bind to another thread's sandbox or token. Without changing the persistent thread-id scheme: - Persist the bound repo (owner/name) in thread metadata on sandbox creation and refuse to reuse a sandbox whose bound repo does not match the current event (SandboxRepoMismatchError); the in-memory proxy also carries the binding. - Bind the GitHub-token cache entries to their repo and evict on a cross-repo read so a colliding thread_id cannot be served another repo's token. - Thread repo through the reviewer and the webhook token resolvers. * fix: scope s3:ListBucket to the releases/ prefix (F-1/IAC-04) The instance role and the GitHub deploy app role granted s3:ListBucket on the whole assets bucket. Every caller (deploy.sh, the publish/rollback scripts) only ever lists under releases/, so add a StringLike s3:prefix=releases/* condition. GetBucketLocation has no s3:prefix in its request context, so it moves to its own unconditioned statement. Also document the accepted F-2 cross-env existence-oracle residual on BatchGetSecretValue. * chore: suppress test-fixture credential false positive; document AUTHZ-002 Add a machine-level suppression for the fake Datadog key in the test_team_credentials encryption-roundtrip fixture (CWE-798, not a real credential). Clarify that the within-org thread-write path is intentional by design (AUTHZ-002) — comment only, no behavior change. * fix: casefold repo-binding keys to avoid spurious cross-repo mismatch GitHub owner/name are case-insensitive. Casefold the owner/name key on both the write (binding) and read (compare) sides — repo_cache_key and the metadata bound_repo read — so Org/Repo and org/repo resolve to one repo and a legitimate same-repo run cannot raise a spurious SandboxRepoMismatchError (Gap 2). * fix: stop leaking upstream auth body in unexpected-result branch The 2xx-but-missing-token/url branch echoed the parsed upstream response body into the user-facing error. Return a generic message and log response_data server-side only, mirroring the existing HTTPStatusError fix (Gap 4). * fix: fail closed for unbound-legacy sandboxes and catch repo mismatch Gap 1: a thread with a persisted sandbox_id but no in-memory cache and no recorded bound_repo (a pre-binding legacy thread, post-deploy) previously reconnected-and-served the sandbox to the current repo, then rebound it. Now fail closed: drop the stale id and recreate a fresh sandbox bound to this repo, logging a reconnect-with-missing-binding event. A sandbox is never served to a repo unless its binding is known and matches; new threads bind on first run unchanged. Gap 3: catch SandboxRepoMismatchError at the agent and reviewer run entrypoints, log it for alarming, and surface a clean sanitized error instead of letting an opaque deep-stack exception crash-loop the worker. * chore: suppress test-fixture credential false positive in token-TTL tests Add a machine-level suppression for the fake "ghp_secret" GitHub token used by the cached-token TTL/revocation unit tests (CWE-798). Not a real credential and not a valid PAT; scoped to the unit test only.
2026-06-29 12:21:19 -04:00
actions: ["s3:ListBucket"],
resources: [`arn:aws:s3:::${p}-assets`],
conditions: { StringLike: { "s3:prefix": ["releases/*"] } },
}),
);
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "GetArtifactBucketLocation",
actions: ["s3:GetBucketLocation"],
resources: [`arn:aws:s3:::${p}-assets`],
}),
);
// Read non-sensitive config from SSM Parameter Store under /open-swe-<env>/*.
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "ReadSsmConfig",
actions: ["ssm:GetParameter", "ssm:GetParameters", "ssm:GetParametersByPath"],
resources: [`arn:aws:ssm:${REGION}:${ACCOUNT}:parameter/${p}/*`],
}),
);
// VALUE access — Secrets Manager under open-swe-<env>/*. Secret ARNs carry a
// random 6-char suffix, hence the trailing `*`. This is the statement that
// actually gates which secret VALUES the box can read: prefix-scoped, so the
// dev box can never read prod secret values (and vice versa). GetSecretValue is
// checked per-secret even when the value is returned via the batch call below.
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "ReadSecretValues",
actions: ["secretsmanager:GetSecretValue", "secretsmanager:DescribeSecret"],
resources: [`arn:aws:secretsmanager:${REGION}:${ACCOUNT}:secret:${p}/*`],
}),
);
// BatchGetSecretValue MUST be granted on `*` — it is a collection action that
// AWS authorizes against the account, NOT the per-secret ARN, REGARDLESS of
// whether the caller uses `--filters` or `--secret-id-list`. A prefix-scoped
// BatchGetSecretValue AccessDenies the whole call ("no identity-based policy
// allows the secretsmanager:BatchGetSecretValue action") — VERIFIED on the live
// dev box 2026-06-29 (the OSWE-IAC-SECRETS-LIST-01 attempt to prefix-scope it
// crash-looped the box once the prior broad grant's eventual-consistency lapsed).
// This `*` does NOT widen VALUE access: a secret value is only returned when the
// prefix-scoped GetSecretValue above also allows it, so cross-env value isolation
// holds. The win that DID survive: fetch-config uses `--secret-id-list` (explicit
// names, no name filter), so `secretsmanager:ListSecrets` is NOT needed and is
// intentionally omitted — the box cannot enumerate secret names account-wide.
fix: resolve security-review findings (sandbox isolation, IAM list scope, webhook replay, info-leak) (#54) * fix: enforce a replay window on Linear webhooks (AUTHZ-001) verify_linear_signature accepted any correctly-signed body with no freshness check, so a captured request could be replayed indefinitely. Parse the signed webhookTimestamp (Unix ms) and reject requests outside a 60s window, failing closed when the field is missing or malformed — mirroring the Slack verifier. * fix: stop leaking upstream auth-error bodies into user comments get_github_token_for_user folded the raw upstream response text into the error string that becomes a Slack/Linear comment (AUTH-RESP-LEAK-01). Log the full body server-side only and return a generic "GitHub auth failed (status <code>)". Also document the accepted shared-installation-token blast radius on the bot-token-only path (AUTHZ-003). * fix: bind sandbox and token caches to repo to prevent thread-id collision A PR head-branch name is attacker-controllable and get_thread_id_from_branch derives a thread_id from its first UUID with no repo binding (TID-COLLIDE-01). The in-memory sandbox cache and the per-thread GitHub-token cache were keyed on thread_id alone, and a cached sandbox was reused after only an echo-ping, so a different repo's webhook could bind to another thread's sandbox or token. Without changing the persistent thread-id scheme: - Persist the bound repo (owner/name) in thread metadata on sandbox creation and refuse to reuse a sandbox whose bound repo does not match the current event (SandboxRepoMismatchError); the in-memory proxy also carries the binding. - Bind the GitHub-token cache entries to their repo and evict on a cross-repo read so a colliding thread_id cannot be served another repo's token. - Thread repo through the reviewer and the webhook token resolvers. * fix: scope s3:ListBucket to the releases/ prefix (F-1/IAC-04) The instance role and the GitHub deploy app role granted s3:ListBucket on the whole assets bucket. Every caller (deploy.sh, the publish/rollback scripts) only ever lists under releases/, so add a StringLike s3:prefix=releases/* condition. GetBucketLocation has no s3:prefix in its request context, so it moves to its own unconditioned statement. Also document the accepted F-2 cross-env existence-oracle residual on BatchGetSecretValue. * chore: suppress test-fixture credential false positive; document AUTHZ-002 Add a machine-level suppression for the fake Datadog key in the test_team_credentials encryption-roundtrip fixture (CWE-798, not a real credential). Clarify that the within-org thread-write path is intentional by design (AUTHZ-002) — comment only, no behavior change. * fix: casefold repo-binding keys to avoid spurious cross-repo mismatch GitHub owner/name are case-insensitive. Casefold the owner/name key on both the write (binding) and read (compare) sides — repo_cache_key and the metadata bound_repo read — so Org/Repo and org/repo resolve to one repo and a legitimate same-repo run cannot raise a spurious SandboxRepoMismatchError (Gap 2). * fix: stop leaking upstream auth body in unexpected-result branch The 2xx-but-missing-token/url branch echoed the parsed upstream response body into the user-facing error. Return a generic message and log response_data server-side only, mirroring the existing HTTPStatusError fix (Gap 4). * fix: fail closed for unbound-legacy sandboxes and catch repo mismatch Gap 1: a thread with a persisted sandbox_id but no in-memory cache and no recorded bound_repo (a pre-binding legacy thread, post-deploy) previously reconnected-and-served the sandbox to the current repo, then rebound it. Now fail closed: drop the stale id and recreate a fresh sandbox bound to this repo, logging a reconnect-with-missing-binding event. A sandbox is never served to a repo unless its binding is known and matches; new threads bind on first run unchanged. Gap 3: catch SandboxRepoMismatchError at the agent and reviewer run entrypoints, log it for alarming, and surface a clean sanitized error instead of letting an opaque deep-stack exception crash-loop the worker. * chore: suppress test-fixture credential false positive in token-TTL tests Add a machine-level suppression for the fake "ghp_secret" GitHub token used by the cached-token TTL/revocation unit tests (CWE-798). Not a real credential and not a valid PAT; scoped to the unit test only.
2026-06-29 12:21:19 -04:00
// F-2 (accepted residual): because the grant is `*`, a caller naming a secret
// in ANOTHER env's prefix learns whether that name EXISTS (an existence oracle
// via the per-secret AccessDenied-vs-not signal) even though the VALUE stays
// gated by the prefix-scoped GetSecretValue above. Accepted within Sea Haven's
// single-tenant account 328440206208 — cross-env VALUE isolation is preserved.
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "BatchGetSecretValues",
actions: ["secretsmanager:BatchGetSecretValue"],
resources: ["*"],
}),
);
// NOTE (T11): SSM SecureString + Secrets Manager here are assumed to use the
// AWS-managed keys (alias/aws/ssm, alias/aws/secretsmanager) for which the
// service grants Decrypt implicitly — so NO kms:Decrypt is granted. If T11
// moves these to a customer CMK, add a scoped `kms:Decrypt` on that key ARN
// ONLY (not `*`).
feat: migrate model providers to Bedrock (Claude) + Fireworks (everything else) (#62) * feat: switch model providers to AWS Bedrock (Claude) and Fireworks (non-Claude) Migrate off direct provider APIs: AWS Bedrock for Anthropic/Claude via the cross-region inference profile us.anthropic.claude-opus-4-8, Fireworks AI for all non-Claude models. Drop OpenAI (gpt-5.5) and Google (gemini-3.5-flash) entirely. DEFAULT_MODEL_ID is now Bedrock Claude; all Fireworks models stay freely selectable for the agent and reviewer graphs and via team/profile defaults. - pyproject: add langchain-aws (ChatBedrockConverse + boto3) - options.py: Bedrock Claude entry + default; remove openai/google entries - model.py: bedrock_converse provider_model_kwargs (effort -> thinking budget), region pin in make_model, bedrock<->fireworks fallback pairing, AWS_REGION/ FIREWORKS_API_KEY local-dev validation - server.py: provider-aware fallback kwargs build - sanitize_thinking_blocks: also sanitize ChatBedrockConverse thinking blocks - model_fallback: treat transient botocore ClientError codes as fallback-worthy - eval_jobs: repoint hardcoded eval model id to Bedrock Claude - tests: repoint dropped model ids; drop obsolete google test module * fix(bedrock): use adaptive thinking + output_config.effort for Opus 4.8 The handoff spec wired Bedrock Converse thinking as {type: enabled, budget_tokens: N}, but Opus 4.7+ rejects that with a ValidationException: thinking.type "enabled" is not supported; it requires thinking.type "adaptive" plus output_config.effort. Verified by live invoke against us.anthropic.claude-opus-4-8 (account 328440206208, us-east-1): the enabled+budget shape 400s, adaptive+effort returns normally. Map profile effort to additional_model_request_fields: {thinking: {type: adaptive, display: summarized}, output_config: {effort: <low|medium|high|xhigh|max>}} reusing anthropic_thinking_for/anthropic_effort_for. Update the two subagent-model tests asserting the old shape. * fix(deploy): seed Bedrock/Fireworks models, not the dropped anthropic:/openai: ids Model selection is store-driven, so seed_store.sh's team_settings/default seed is what runs in prod. It still seeded the removed providers, which would fail at runtime after the migration: - agent/builder: anthropic:claude-opus-4-8 -> bedrock_converse:us.anthropic.claude-opus-4-8 - reviewer: openai:gpt-5.5 (dropped) -> bedrock_converse:us.anthropic.claude-opus-4-8 (set SEED_REVIEWER_MODEL to a Fireworks model for a cross-family reviewer) - fetch-config REQUIRED_PROVIDER_KEYS default ANTHROPIC_API_KEY,OPENAI_API_KEY -> FIREWORKS_API_KEY (Bedrock auths via host IAM role; dropping the old keys would otherwise fail-fast at boot) - docs (DEPLOYMENT/ROTATION/put-config) updated to match. Surfaced by the cross-family review + verified against deploy/. * fix(bedrock): security-review NITs — region resolution, error sanitization, reasoning-block strip From /sh-security-review (all confirmed-low): - model.py: resolve region from AWS_REGION OR AWS_DEFAULT_REGION (matches validate_local_dev_llm_config) so the validated region is the one actually used. - model_fallback.py: sanitize Bedrock AccessDenied/ResourceNotFound errors to the error code only, so the role ARN + account id in the raw botocore message never reach logs or the user channel (CWE-209). - sanitize_thinking_blocks.py: also strip empty Bedrock reasoning_content blocks (Converse emits reasoning_content, not thinking) so the middleware is not a no-op on Bedrock; + unit tests. (Empty blocks replay fine today; defensive.) * deploy(bedrock): grant instance-role Bedrock invoke + repoint LLM_MODEL_ID / eval model ids Deployment-readiness for the Bedrock migration (PR #62): - instance-role.ts: least-privilege bedrock:InvokeModel[WithResponseStream] on the us.anthropic.claude-opus-4-8 inference-profile ARN + the foundation-model ARN in each routed region (us-east-1/2, us-west-2). The model runs in the server process on the box, so the EC2 instance role is the principal. Simulator-verified (allowed for opus-4-8, implicitDeny for other models) and synth-verified. Passed the mandatory GPT-4.1 IAM cross-review (no blockers, least-privilege confirmed). - config-store.ts: IaC SSM LLM_MODEL_ID anthropic:claude-opus-4-8 -> bedrock_converse:us.anthropic.claude-opus-4-8. This SSM value overrides seed_store.sh's default via pick precedence, so the seed-script fix alone was insufficient — both sources now point at the supported Bedrock id. - infra/README.md + evals/reviewer/config.toml: repoint stale anthropic:/google_genai: ids to the Bedrock id (config.toml's model_id was an active, now-broken value). AWS_REGION is already wired via user-data.sh (IMDS -> boot.env), so no change needed there. * chore(secrets): drop OPENAI/GOOGLE/GROQ key shells (revoked, providers removed) Those three providers were dropped in the Bedrock/Fireworks migration and their keys revoked; the live Secrets Manager objects (open-swe-{dev,prod}/{OPENAI,GOOGLE,GROQ}_API_KEY) were deleted (7-day recovery). Remove them from the IaC so a future cdk deploy does not recreate the shells, and from fetch-config's mirror array so boot stops requesting them: - config-store.ts SECRET_VARS + descriptions (28 -> 25 shells) - fetch-config.sh SECRET_VARS array (kept in lockstep) - put-config.sh: drop the put_secret lines; ANTHROPIC_API_KEY re-labelled optional (eval judge only — Bedrock builder/reviewer auth via the host IAM role). REQUIRED_PROVIDER_KEYS is not set in SSM, so it uses the FIREWORKS_API_KEY default.
2026-06-29 15:57:19 -04:00
// Invoke the Bedrock Claude model. DEFAULT_MODEL_ID is
// `bedrock_converse:us.anthropic.claude-opus-4-8`, and the model runs in the
// LangGraph server PROCESS on this box (not in the sandbox), so the EC2
// instance role is the calling principal. The `us.` cross-region inference
// profile fans out to us-east-1 / us-east-2 / us-west-2, and Bedrock authorizes
// InvokeModel against BOTH the inference-profile ARN AND the underlying
// foundation-model ARN in each routed region — all four resources are required
// or the call AccessDenies. Scoped to opus-4-8 ONLY (least-privilege): adding a
// new Bedrock model to SUPPORTED_MODELS means extending this resource list.
// IAM change — flag for T4 (GPT-4.1 IAM cross-review) / T5 (/sh-security-review).
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "InvokeBedrockClaude",
actions: ["bedrock:InvokeModel", "bedrock:InvokeModelWithResponseStream"],
resources: [
`arn:aws:bedrock:${REGION}:${ACCOUNT}:inference-profile/us.anthropic.claude-opus-4-8`,
"arn:aws:bedrock:us-east-1::foundation-model/anthropic.claude-opus-4-8",
"arn:aws:bedrock:us-east-2::foundation-model/anthropic.claude-opus-4-8",
"arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-opus-4-8",
],
}),
);
// Ship application logs to CloudWatch Logs under /open-swe/<env>/*.
this.role.addToPolicy(
new iam.PolicyStatement({
sid: "PutAppLogs",
actions: [
"logs:CreateLogGroup",
"logs:CreateLogStream",
"logs:PutLogEvents",
"logs:DescribeLogStreams",
],
resources: [
`arn:aws:logs:${REGION}:${ACCOUNT}:log-group:/open-swe/${env}/*`,
`arn:aws:logs:${REGION}:${ACCOUNT}:log-group:/open-swe/${env}/*:*`,
],
}),
);
}
}