chore(infra): remove API Gateway and Lambda dual-run (PLAT-216) (#264)

Origins already point at the Fargate hostnames. Drop the HTTP API, eight
functions, zip CD, and Lambda/API Gateway alarms while keeping leftover
Lambda IAM so Paychex can still name weekly-post.
This commit is contained in:
Adam Moussa 2026-09-21 22:37:13 +00:00 • committed by GitHub
parent 6c4018d6b8
commit cbda46ba87
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
15 changed files with 58 additions and 663 deletions

View file

@ -9,7 +9,7 @@ name: Deploy API
# workflow_dispatch -> chosen environment at a chosen ref
#
# Releases are cut by a human with `gh release create vX.Y.Z --target main`.
# Nothing here creates an HCP run. Zip CD stays in deploy.yaml until cutover.
# Nothing here creates an HCP run.
on:
push:
@ -18,7 +18,6 @@ on:
- "terraform/**"
- "docs/**"
- "*.md"
- ".github/workflows/deploy.yaml"
- ".github/workflows/ci.yaml"
- ".github/workflows/changelog-guard.yml"
- ".github/workflows/labeler.yml"

View file

@ -1,154 +0,0 @@
name: Deploy
# Terraform owns Lambda skeletons. This workflow ships zips to prod and calls
# update-function-code. It never creates an HCP run. No GitHub Releases and no
# tagging in this workflow.
on:
push:
branches: [main]
paths-ignore:
- "terraform/**"
- "docs/**"
- "README.md"
- "SETUP.md"
- "AGENTS.md"
workflow_dispatch:
inputs:
ref:
description: "Git ref to build and deploy (tag, branch, or SHA). Empty means the workflow ref."
required: false
type: string
default: ""
permissions:
contents: read
jobs:
deploy:
name: Deploy to prod
runs-on: ubuntu-latest
timeout-minutes: 30
environment: prod
concurrency:
group: deploy-afterhours-prod
cancel-in-progress: false
permissions:
contents: read
id-token: write
env:
AWS_REGION: us-east-1
DEPLOY_ROLE_ARN: ${{ vars.DEPLOY_ROLE_ARN }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ github.event_name == 'workflow_dispatch' && inputs.ref || github.sha }}
persist-credentials: false
- name: Resolve commit
id: commit
run: |
set -euo pipefail
sha="$(git rev-parse HEAD)"
echo "sha=${sha}" >> "$GITHUB_OUTPUT"
echo "Building ${sha}"
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
with:
python-version: "3.12"
- name: Build function zips
env:
GIT_SHA: ${{ steps.commit.outputs.sha }}
run: |
set -euo pipefail
python scripts/package_lambdas.py --git-sha "${GIT_SHA}" --out-dir build/packages
python - <<'PY'
import os, zipfile
from pathlib import Path
sha = os.environ["GIT_SHA"]
names = [
"slack_bot",
"weekly_post",
"roster_sync",
"roster_api",
"ring_scheduler",
"holiday_router",
"release_notifier",
"portal_api",
]
for name in names:
path = Path("build/packages") / f"{name}.zip"
if not path.is_file():
raise SystemExit(f"missing {path}")
with zipfile.ZipFile(path) as zf:
info = zf.read("shared/build_info.py").decode()
if sha not in info:
raise SystemExit(f"{path} missing GIT_SHA {sha}")
if "shared/sentry_init.py" not in zf.namelist():
raise SystemExit(f"{path} missing bundled shared package")
print("zips ok")
PY
- name: Configure AWS credentials using OIDC
uses: aws-actions/configure-aws-credentials@e1253824e5c10ff9df46874f81ed3ec929e19cfd # v6.3.0
with:
role-to-assume: ${{ env.DEPLOY_ROLE_ARN }}
aws-region: us-east-1
audience: sts.amazonaws.com
- name: Get deploy parameters
id: deploy
run: |
set -euo pipefail
prefix=/afterhours-shift-manager/deploy
ARTIFACTS_BUCKET=$(aws ssm get-parameter --name "${prefix}/artifacts-bucket" --query Parameter.Value --output text)
{
echo "artifacts_bucket=${ARTIFACTS_BUCKET}"
echo "slack_bot=$(aws ssm get-parameter --name "${prefix}/slack_bot-function-name" --query Parameter.Value --output text)"
echo "weekly_post=$(aws ssm get-parameter --name "${prefix}/weekly_post-function-name" --query Parameter.Value --output text)"
echo "roster_sync=$(aws ssm get-parameter --name "${prefix}/roster_sync-function-name" --query Parameter.Value --output text)"
echo "roster_api=$(aws ssm get-parameter --name "${prefix}/roster_api-function-name" --query Parameter.Value --output text)"
echo "ring_scheduler=$(aws ssm get-parameter --name "${prefix}/ring_scheduler-function-name" --query Parameter.Value --output text)"
echo "holiday_router=$(aws ssm get-parameter --name "${prefix}/holiday_router-function-name" --query Parameter.Value --output text)"
echo "release_notifier=$(aws ssm get-parameter --name "${prefix}/release_notifier-function-name" --query Parameter.Value --output text)"
echo "portal_api=$(aws ssm get-parameter --name "${prefix}/portal_api-function-name" --query Parameter.Value --output text)"
} >> "${GITHUB_OUTPUT}"
- name: Upload zips and update function code
env:
ARTIFACTS_BUCKET: ${{ steps.deploy.outputs.artifacts_bucket }}
GIT_SHA: ${{ steps.commit.outputs.sha }}
SLACK_BOT: ${{ steps.deploy.outputs.slack_bot }}
WEEKLY_POST: ${{ steps.deploy.outputs.weekly_post }}
ROSTER_SYNC: ${{ steps.deploy.outputs.roster_sync }}
ROSTER_API: ${{ steps.deploy.outputs.roster_api }}
RING_SCHEDULER: ${{ steps.deploy.outputs.ring_scheduler }}
HOLIDAY_ROUTER: ${{ steps.deploy.outputs.holiday_router }}
RELEASE_NOTIFIER: ${{ steps.deploy.outputs.release_notifier }}
PORTAL_API: ${{ steps.deploy.outputs.portal_api }}
run: |
set -euo pipefail
keys=(
slack_bot:"${SLACK_BOT}"
weekly_post:"${WEEKLY_POST}"
roster_sync:"${ROSTER_SYNC}"
roster_api:"${ROSTER_API}"
ring_scheduler:"${RING_SCHEDULER}"
holiday_router:"${HOLIDAY_ROUTER}"
release_notifier:"${RELEASE_NOTIFIER}"
portal_api:"${PORTAL_API}"
)
for pair in "${keys[@]}"; do
name="${pair%%:*}"
fn="${pair#*:}"
key="functions/${name}/${GIT_SHA}.zip"
aws s3 cp "build/packages/${name}.zip" "s3://${ARTIFACTS_BUCKET}/${key}"
aws lambda update-function-code \
--function-name "${fn}" \
--s3-bucket "${ARTIFACTS_BUCKET}" \
--s3-key "${key}" \
--query '{Function:FunctionName,Sha256:CodeSha256,Updated:LastModified}' \
--output table
aws lambda wait function-updated-v2 --function-name "${fn}"
done

View file

@ -60,7 +60,7 @@ Example: `/oncall admin holiday add 2026-07-04 2 x2 Independence Day` schedules
- **Runtime**: Python 3.12 on ECS Fargate (arm64) plus dual-run Lambdas until cutover. seahaven-prod `011934824531`, seahaven-dev `710827005802`
- **Data**: DynamoDB single-table (`afterhours-shifts`)
- **IaC**: HCP Terraform workspaces tagged `app:afterhours-shift-manager` (`afterhours-shift-manager-dev` / `-prod`) plus GitHub Actions `deploy.yaml` (zips) and `deploy-api.yaml` (image). Terraform does not package `src/`.
- **IaC**: HCP Terraform workspaces tagged `app:afterhours-shift-manager` (`afterhours-shift-manager-dev` / `-prod`) plus GitHub Actions `deploy-api.yaml` (image). Terraform does not package `src/`.
- **Slack**: Slack Bolt on `POST /slack/events`
- **3CX Integration**: Queue routing updated directly via 3CX Queue XAPI
- **Secrets**: AWS Secrets Manager (`afterhours-shift-manager/*`)
@ -233,27 +233,22 @@ The canonical map of Sea Haven's AWS infrastructure lives in Confluence. This pr
## Deployment
Infrastructure is applied by HCP Terraform workspace `afterhours-shift-manager-prod` (VCS on `main`, working directory `terraform/`, file trigger `terraform/**` only). Function code is shipped by `.github/workflows/deploy.yaml` on push to `main` (`environment: prod`). A terraform-only merge does not run the zip deploy. A mixed app+terraform merge may race the apply; re-run the deploy job if the functions are still stubs.
Manual zip redeploy: Actions → Deploy → Run workflow (`workflow_dispatch`, always prod). Do not `terraform apply` locally to prod.
Infrastructure is applied by HCP Terraform workspace `afterhours-shift-manager-prod` (VCS on `main`, working directory `terraform/`, file trigger `terraform/**` only). The Flask image is shipped by `.github/workflows/deploy-api.yaml`.
## Monitoring & Alarms
All CloudWatch alarms are defined in `terraform/alarms.tf` and notify the shared
`site-alerts` SNS topic (→ AWS Chatbot → Slack). None set `OKActions` — recovery
is not paged. Alarm names follow `Lambda-<Metric>-<fn>`
(e.g. `Lambda-Errors-afterhours-ring-scheduler`).
is not paged. Alarm names follow `ALB-<Metric>-afterhours-shift-manager` and
`DDB-<Metric>-afterhours-shifts`.
**Lambda alarms** (all eight functions: `afterhours-shift-manager`,
`afterhours-weekly-post`, `afterhours-roster-sync`, `afterhours-roster-api`,
`afterhours-ring-scheduler`, `afterhours-holiday-router`,
`afterhours-release-notifier`, `afterhours-portal-api`):
**ALB alarms**:
| Alarm | Metric | Condition | Notes |
|---|---|---|---|
| `Lambda-Errors-<fn>` | `Errors` (Sum) | `>= 1` over one 5-min period | `TreatMissingData: notBreaching` |
| `Lambda-Duration-<fn>` | `Duration` (Maximum, ms) | `>= ~80% of timeout`, 2 of 3 5-min periods | Thresholds: 24000 ms (30s-timeout fns) / 48000 ms (60s-timeout fns) |
| `Lambda-Throttles-<fn>` | `Throttles` (Sum) | `>= 1` over one 5-min period | `TreatMissingData: notBreaching` |
| Alarm | Metric | Condition |
|---|---|---|
| `ALB-5xx-afterhours-shift-manager` | `HTTPCode_Target_5XX_Count` (Sum) | `> 0` over one 5-min period |
| `ALB-Latency-afterhours-shift-manager` | `TargetResponseTime` (p99, s) | `>= 3` s, 2 of 3 5-min periods |
| `ALB-UnhealthyHost-afterhours-shift-manager` | `UnHealthyHostCount` (Maximum) | `> 0`, 3 of 3 1-min periods |
**DynamoDB alarm** (`afterhours-shifts` table):
@ -268,16 +263,7 @@ transition normally. `ThrottledRequests` and `SystemErrors` are intentionally
**not** alarmed: AWS emits them only at `TableName`+`Operation` granularity, so a
`TableName`-only alarm would sit permanently in `INSUFFICIENT_DATA`.
**API Gateway alarms** (HTTP API, `AWS/ApiGateway` v2 metrics, `ApiId` dimension):
| Alarm | Metric | Condition |
|---|---|---|
| `ApiGateway-4xx-<apiId>` | `4xx` (Sum) | `>= 5` over one 5-min period |
| `ApiGateway-5xx-<apiId>` | `5xx` (Sum) | `>= 1` over one 5-min period |
| `ApiGateway-Latency-<apiId>` | `Latency` (p99, ms) | `>= 3000` ms, 2 of 3 5-min periods |
> Duration and API latency thresholds are starting points and may be tuned after
> observing real traffic.
**API Gateway alarms** were removed with the Fargate cutover (PLAT-216).
## Releases & Versioning
@ -286,8 +272,8 @@ The bot is versioned with SemVer, driven entirely by **`CHANGELOG.md`**. The
`python scripts/sync_changelog.py` after editing the root file. Changelog Guard
enforces that the in-package copy matches.
GitHub Releases and git tags are not cut by `deploy.yaml`. `afterhours-release-notifier`
exists as a function skeleton; CD does not invoke it until tagging exists.
GitHub Releases and git tags are not cut by `deploy-api.yaml`. `afterhours-release-notifier`
exists as leftover IAM from dual-run; CD does not invoke it.
## Testing

View file

@ -187,24 +187,12 @@ is copied.
## 9. Fargate cutover (PLAT-216)
Dual-run ECS beside API Gateway. Do not dual-write 3CX. Do not flip Slack
without the `afterhours-shift-manager-dev` workspace already serving
`/api/health`.
1. Image deploy via `deploy-api.yaml`. `GET /api/health` reports the real sha.
2. Recreate outstanding `holiday-activate-*` / `holiday-deactivate-*` onto
jobs SQS (same class of work as `recreate_holiday_schedules.py`):
`python scripts/cutover/retarget_holiday_schedules_to_sqs.py --profile prod --queue-arn <jobs_queue_arn>`
then `--execute`.
3. Instant cut: Slack Request URL, Paychex `AFTERHOURS_BASE_URL`, portal
`VITE_SHIFTS_API_BASE` → `https://afterhours.seahaven.com` (dev hostname in
portal-dev). Smoke `/oncall`, roster PUT/DELETE, `GET /api/shifts`, one
holiday GetSchedule, ring job.
4. Set `ecs_schedules_enabled=true` and keep `schedules_enabled=false`. Confirm
no 3CX writers remain on Lambda.
5. After smoke, remove API Gateway, the eight Lambdas, zip packaging, and
Lambda alarms in a follow-up apply. ALB 5xx/latency and unhealthy-host
alarms stay.
Completed 2026-09-21. Live path is ECS Fargate behind
`https://afterhours.seahaven.com` (dev: `https://afterhours.dev.seahaven.com`).
Slack Request URL, Paychex `AFTERHOURS_BASE_URL`, and portal `VITE_SHIFTS_API_BASE`
point at those hosts. Jobs use EventBridge Scheduler → SQS (`ecs_schedules_enabled=true`,
Lambda EventBridge `schedules_enabled=false`). Public DNS is out of band in the
mgmt `seahaven.com` zone and the external-dev `dev.seahaven.com` zone.
## Commands Reference

View file

@ -1,83 +1,3 @@
locals {
lambda_alarm_matrix = {
errors = {
metric_name = "Errors"
statistic = "Sum"
evaluation_periods = 1
datapoints_to_alarm = 1
threshold = 1
comparison = "GreaterThanOrEqualToThreshold"
period = 300
}
throttles = {
metric_name = "Throttles"
statistic = "Sum"
evaluation_periods = 1
datapoints_to_alarm = 1
threshold = 1
comparison = "GreaterThanOrEqualToThreshold"
period = 300
}
}
lambda_alarms = {
for pair in flatten([
for fn_key, fn in local.functions : [
for metric_key, metric in local.lambda_alarm_matrix : {
key = "${fn_key}-${metric_key}"
fn_key = fn_key
function = fn.function_name
metric_key = metric_key
metric_name = metric.metric_name
statistic = metric.statistic
evaluation = metric.evaluation_periods
datapoints = metric.datapoints_to_alarm
threshold = metric.threshold
comparison = metric.comparison
period = metric.period
description = metric_key == "errors" ? "${fn.function_name} reported one or more errors" : "${fn.function_name} was throttled (concurrency limit hit)"
}
]
]) : pair.key => pair
}
}
resource "aws_cloudwatch_metric_alarm" "lambda_errors_throttles" {
for_each = local.lambda_alarms
alarm_name = "Lambda-${title(each.value.metric_key)}-${each.value.function}"
alarm_description = each.value.description
namespace = "AWS/Lambda"
metric_name = each.value.metric_name
dimensions = { FunctionName = each.value.function }
statistic = each.value.statistic
period = each.value.period
evaluation_periods = each.value.evaluation
datapoints_to_alarm = each.value.datapoints
threshold = each.value.threshold
comparison_operator = each.value.comparison
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "lambda_duration" {
for_each = local.functions
alarm_name = "Lambda-Duration-${each.value.function_name}"
alarm_description = "${each.value.function_name} duration approaching its ${each.value.timeout}s timeout (>=${each.value.duration_ms}ms)"
namespace = "AWS/Lambda"
metric_name = "Duration"
dimensions = { FunctionName = each.value.function_name }
statistic = "Maximum"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = each.value.duration_ms
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "ddb_read_throttle" {
alarm_name = "DDB-ReadThrottle-${local.table_name}"
alarm_description = "afterhours-shifts table had one or more read throttle events"
@ -108,52 +28,6 @@ resource "aws_cloudwatch_metric_alarm" "ddb_write_throttle" {
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "api_4xx" {
alarm_name = "ApiGateway-4xx-${aws_apigatewayv2_api.http.id}"
alarm_description = "Elevated 4xx responses on the afterhours HTTP API"
namespace = "AWS/ApiGateway"
metric_name = "4xx"
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 5
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "api_5xx" {
alarm_name = "ApiGateway-5xx-${aws_apigatewayv2_api.http.id}"
alarm_description = "5xx responses on the afterhours HTTP API"
namespace = "AWS/ApiGateway"
metric_name = "5xx"
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 1
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "api_latency" {
alarm_name = "ApiGateway-Latency-${aws_apigatewayv2_api.http.id}"
alarm_description = "p99 latency on the afterhours HTTP API exceeded 3s"
namespace = "AWS/ApiGateway"
metric_name = "Latency"
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
extended_statistic = "p99"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = 3000
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "alb_5xx" {
alarm_name = "ALB-5xx-${local.project}"
alarm_description = "ALB 5xx from afterhours-shift-manager"

View file

@ -1,126 +0,0 @@
# HTTP API: Slack events, Paychex roster contract, and portal shift API.
resource "aws_apigatewayv2_api" "http" {
name = local.project
protocol_type = "HTTP"
description = "afterhours-shift-manager Slack, roster, and portal API"
cors_configuration {
allow_origins = [
"https://internal.seahaven.com",
"https://internal.dev.seahaven.com",
"http://localhost:5173",
"http://localhost:4173",
]
allow_methods = ["GET", "POST", "DELETE", "OPTIONS"]
allow_headers = ["Authorization", "Content-Type"]
allow_credentials = false
max_age = 3600
}
}
resource "aws_apigatewayv2_integration" "slack_bot" {
api_id = aws_apigatewayv2_api.http.id
integration_type = "AWS_PROXY"
integration_method = "POST"
integration_uri = aws_lambda_function.this["slack_bot"].invoke_arn
payload_format_version = "2.0"
timeout_milliseconds = 30000
}
resource "aws_apigatewayv2_integration" "roster_api" {
api_id = aws_apigatewayv2_api.http.id
integration_type = "AWS_PROXY"
integration_method = "POST"
integration_uri = aws_lambda_function.this["roster_api"].invoke_arn
payload_format_version = "2.0"
timeout_milliseconds = 30000
}
resource "aws_apigatewayv2_integration" "portal_api" {
api_id = aws_apigatewayv2_api.http.id
integration_type = "AWS_PROXY"
integration_method = "POST"
integration_uri = aws_lambda_function.this["portal_api"].invoke_arn
payload_format_version = "2.0"
timeout_milliseconds = 30000
}
resource "aws_apigatewayv2_route" "slack_events" {
api_id = aws_apigatewayv2_api.http.id
route_key = "POST /slack/events"
target = "integrations/${aws_apigatewayv2_integration.slack_bot.id}"
}
resource "aws_apigatewayv2_route" "put_roster" {
api_id = aws_apigatewayv2_api.http.id
route_key = "PUT /roster"
target = "integrations/${aws_apigatewayv2_integration.roster_api.id}"
}
resource "aws_apigatewayv2_route" "delete_roster" {
api_id = aws_apigatewayv2_api.http.id
route_key = "DELETE /roster/{extension}"
target = "integrations/${aws_apigatewayv2_integration.roster_api.id}"
}
resource "aws_apigatewayv2_route" "portal_shifts" {
api_id = aws_apigatewayv2_api.http.id
route_key = "ANY /api/shifts"
target = "integrations/${aws_apigatewayv2_integration.portal_api.id}"
}
resource "aws_apigatewayv2_route" "portal_shifts_proxy" {
api_id = aws_apigatewayv2_api.http.id
route_key = "ANY /api/shifts/{proxy+}"
target = "integrations/${aws_apigatewayv2_integration.portal_api.id}"
}
resource "aws_apigatewayv2_stage" "default" {
api_id = aws_apigatewayv2_api.http.id
name = "$default"
auto_deploy = true
access_log_settings {
destination_arn = aws_cloudwatch_log_group.api_access.arn
format = "{\"requestId\":\"$context.requestId\",\"ip\":\"$context.identity.sourceIp\",\"requestTime\":\"$context.requestTime\",\"method\":\"$context.httpMethod\",\"routeKey\":\"$context.routeKey\",\"status\":\"$context.status\",\"protocol\":\"$context.protocol\",\"responseLength\":\"$context.responseLength\",\"integrationError\":\"$context.integrationErrorMessage\"}"
}
default_route_settings {
throttling_burst_limit = 50
throttling_rate_limit = 100
}
depends_on = [
aws_apigatewayv2_route.slack_events,
aws_apigatewayv2_route.put_roster,
aws_apigatewayv2_route.delete_roster,
aws_apigatewayv2_route.portal_shifts,
aws_apigatewayv2_route.portal_shifts_proxy,
aws_iam_role_policy.hcptf_apply_services,
]
}
resource "aws_lambda_permission" "api_slack_bot" {
statement_id = "AllowApiGatewayInvokeSlackBot"
action = "lambda:InvokeFunction"
function_name = aws_lambda_function.this["slack_bot"].function_name
principal = "apigateway.amazonaws.com"
source_arn = "${aws_apigatewayv2_api.http.execution_arn}/*/*"
}
resource "aws_lambda_permission" "api_roster_api" {
statement_id = "AllowApiGatewayInvokeRosterApi"
action = "lambda:InvokeFunction"
function_name = aws_lambda_function.this["roster_api"].function_name
principal = "apigateway.amazonaws.com"
source_arn = "${aws_apigatewayv2_api.http.execution_arn}/*/*"
}
resource "aws_lambda_permission" "api_portal_api" {
statement_id = "AllowApiGatewayInvokePortalApi"
action = "lambda:InvokeFunction"
function_name = aws_lambda_function.this["portal_api"].function_name
principal = "apigateway.amazonaws.com"
source_arn = "${aws_apigatewayv2_api.http.execution_arn}/*/*"
}

View file

@ -186,7 +186,6 @@ locals {
{ name = "TCX_SECRET_PREFIX", value = "afterhours-shift-manager/3cx-" },
{ name = "QUEUE_NUMBER", value = var.queue_number },
{ name = "TZ", value = var.timezone },
{ name = "HOLIDAY_ROUTER_ARN", value = local.holiday_router_arn },
{ name = "HOLIDAY_SCHEDULER_ROLE_ARN", value = local.holiday_scheduler_role_arn },
{ name = "HOLIDAY_SCHEDULE_GROUP", value = "default" },
{ name = "SENTRY_DSN", value = var.sentry_dsn },

View file

@ -1,76 +0,0 @@
# EventBridge schedules. Every schedule is an EST/EDT pair firing the same
# function one hour apart in UTC: EventBridge cron has no timezone. Both fire
# year-round and the handlers are idempotent. Keep schedules_enabled=false
# until Slack and Paychex point at this stack.
locals {
schedules = {
weekly-post-est = {
description = "Post weekly schedule Monday 7am EST"
schedule = "cron(0 12 ? * MON *)"
function_key = "weekly_post"
}
weekly-post-edt = {
description = "Post weekly schedule Monday 7am EDT"
schedule = "cron(0 11 ? * MON *)"
function_key = "weekly_post"
}
roster-sync-est = {
description = "Sync roster from 3CX at 6am EST"
schedule = "cron(0 11 ? * * *)"
function_key = "roster_sync"
}
roster-sync-edt = {
description = "Sync roster from 3CX at 6am EDT"
schedule = "cron(0 10 ? * * *)"
function_key = "roster_sync"
}
ring-scheduler-daily-est = {
description = "Update 3CX queue at 8am EST"
schedule = "cron(0 13 ? * * *)"
function_key = "ring_scheduler"
}
ring-scheduler-daily-edt = {
description = "Update 3CX queue at 8am EDT"
schedule = "cron(0 12 ? * * *)"
function_key = "ring_scheduler"
}
ring-scheduler-weekend-est = {
description = "Update 3CX queue at 5pm EST weekends"
schedule = "cron(0 22 ? * SAT,SUN *)"
function_key = "ring_scheduler"
}
ring-scheduler-weekend-edt = {
description = "Update 3CX queue at 5pm EDT weekends"
schedule = "cron(0 21 ? * SAT,SUN *)"
function_key = "ring_scheduler"
}
}
}
resource "aws_cloudwatch_event_rule" "schedule" {
for_each = local.schedules
name = "${local.project}-${each.key}"
description = each.value.description
schedule_expression = each.value.schedule
state = var.schedules_enabled ? "ENABLED" : "DISABLED"
}
resource "aws_cloudwatch_event_target" "schedule" {
for_each = local.schedules
rule = aws_cloudwatch_event_rule.schedule[each.key].name
target_id = "${local.project}-${each.key}"
arn = aws_lambda_function.this[each.value.function_key].arn
}
resource "aws_lambda_permission" "schedule" {
for_each = local.schedules
statement_id = "AllowEventBridgeInvoke-${each.key}"
action = "lambda:InvokeFunction"
function_name = aws_lambda_function.this[each.value.function_key].function_name
principal = "events.amazonaws.com"
source_arn = aws_cloudwatch_event_rule.schedule[each.key].arn
}

View file

@ -1,8 +1,8 @@
# GitHub Actions OIDC role for .github/workflows/deploy.yaml and deploy-api.yaml.
# GitHub Actions OIDC role for .github/workflows/deploy-api.yaml.
#
# One role: GitHub Environments have a single DEPLOY_ROLE_ARN. Trust is pinned
# to Environments dev and prod (immutable and classic subject forms) and
# job_workflow_ref to deploy.yaml at main plus deploy-api.yaml at main and v*.
# job_workflow_ref to deploy-api.yaml at main and v*.
data "aws_iam_policy_document" "github_deploy_assume" {
statement {
@ -31,7 +31,6 @@ data "aws_iam_policy_document" "github_deploy_assume" {
test = "StringLike"
variable = "token.actions.githubusercontent.com:job_workflow_ref"
values = [
"${var.github_repo}/.github/workflows/deploy.yaml@refs/heads/${var.github_deploy_branch}",
"${var.github_repo}/.github/workflows/deploy-api.yaml@refs/heads/${var.github_deploy_branch}",
"${var.github_repo}/.github/workflows/deploy-api.yaml@refs/tags/v*",
]
@ -42,43 +41,12 @@ data "aws_iam_policy_document" "github_deploy_assume" {
resource "aws_iam_role" "github_deploy" {
name = local.deploy_role
path = "/tf-managed/"
description = "GitHub Actions zip and image deploy role for ${var.github_repo}"
description = "GitHub Actions image deploy role for ${var.github_repo}"
assume_role_policy = data.aws_iam_policy_document.github_deploy_assume.json
max_session_duration = 3600
}
data "aws_iam_policy_document" "github_deploy" {
statement {
sid = "ListArtifactsBucket"
effect = "Allow"
actions = [
"s3:GetBucketLocation",
"s3:ListBucket",
]
resources = [aws_s3_bucket.artifacts.arn]
}
statement {
sid = "UploadFunctionArtifacts"
effect = "Allow"
actions = [
"s3:GetObject",
"s3:PutObject",
]
resources = ["${aws_s3_bucket.artifacts.arn}/functions/*"]
}
statement {
sid = "UpdateFunctionCode"
effect = "Allow"
actions = [
"lambda:GetFunction",
"lambda:GetFunctionConfiguration",
"lambda:UpdateFunctionCode",
]
resources = [for fn in local.functions : "arn:aws:lambda:${var.aws_region}:${local.account_id}:function:${fn.function_name}"]
}
statement {
sid = "EcrAuth"
effect = "Allow"

View file

@ -1,9 +1,6 @@
# Terraform owns the function skeletons (role, runtime, memory, environment).
# Code is owned by .github/workflows/deploy.yaml, which uploads
# functions/<name>/<sha>.zip and calls update-function-code. The lifecycle
# block is the seam: an app deploy is not drift, and a Terraform apply never
# rolls the code back to the bootstrap stub. GIT_SHA is written into
# shared/build_info.py at zip time, not set here.
# Leftover Lambda IAM roles from dual-run. The eight functions are gone;
# weekly-post remains so paychex-checkcomponents can keep that principal
# until a follow-up queue-policy apply drops it.
data "aws_iam_policy_document" "lambda_assume" {
statement {
@ -306,33 +303,3 @@ resource "aws_iam_role_policy_attachment" "lambda_basic" {
role = aws_iam_role.lambda[each.key].name
policy_arn = "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole"
}
resource "aws_lambda_function" "this" {
for_each = local.functions
function_name = each.value.function_name
role = aws_iam_role.lambda[each.key].arn
handler = each.value.handler
runtime = "python3.12"
architectures = ["arm64"]
memory_size = 1024
timeout = each.value.timeout
s3_bucket = aws_s3_bucket.artifacts.id
s3_key = aws_s3_object.bootstrap_stub.key
source_code_hash = data.archive_file.bootstrap_stub.output_base64sha256
environment {
variables = local.lambda_env[each.key]
}
lifecycle {
ignore_changes = [filename, s3_bucket, s3_key, s3_object_version, source_code_hash]
}
depends_on = [
aws_cloudwatch_log_group.lambda,
aws_iam_role_policy.lambda,
aws_iam_role_policy_attachment.lambda_basic,
]
}

View file

@ -1,15 +1,3 @@
resource "aws_cloudwatch_log_group" "lambda" {
for_each = local.functions
name = "/aws/lambda/${each.value.function_name}"
retention_in_days = 60
}
resource "aws_cloudwatch_log_group" "api_access" {
name = "/aws/apigateway/${local.project}"
retention_in_days = 90
}
resource "aws_cloudwatch_log_group" "api" {
name = "/ecs/${local.project}"
retention_in_days = local.is_prod ? 60 : 14

View file

@ -1,11 +1,11 @@
output "slack_request_url" {
description = "Slack app Request URL (slash command and interactivity)."
value = "${aws_apigatewayv2_api.http.api_endpoint}/slack/events"
value = "${local.api_url}/slack/events"
}
output "api_origin" {
description = "HTTP API origin for Paychex AFTERHOURS_BASE_URL. No /roster suffix."
value = aws_apigatewayv2_api.http.api_endpoint
description = "HTTP origin for Paychex AFTERHOURS_BASE_URL and portal VITE_SHIFTS_API_BASE. No /roster suffix."
value = local.api_url
}
output "shift_table_name" {
@ -14,17 +14,17 @@ output "shift_table_name" {
}
output "holiday_scheduler_role_arn" {
description = "Role EventBridge Scheduler assumes to invoke the holiday router."
description = "Role EventBridge Scheduler assumes to enqueue holiday jobs."
value = aws_iam_role.holiday_scheduler.arn
}
output "github_deploy_role_arn" {
description = "OIDC role ARN for deploy.yaml and deploy-api.yaml (GitHub Environment variable DEPLOY_ROLE_ARN)."
description = "OIDC role ARN for deploy-api.yaml (GitHub Environment variable DEPLOY_ROLE_ARN)."
value = aws_iam_role.github_deploy.arn
}
output "artifacts_bucket_name" {
description = "Lambda artifacts bucket. deploy.yaml uploads functions/<name>/<sha>.zip."
description = "Leftover Lambda artifacts bucket from dual-run zip CD."
value = aws_s3_bucket.artifacts.id
}

View file

@ -36,13 +36,6 @@ resource "aws_iam_role" "holiday_scheduler" {
}
data "aws_iam_policy_document" "holiday_scheduler" {
statement {
sid = "InvokeHolidayRouter"
effect = "Allow"
actions = ["lambda:InvokeFunction"]
resources = [local.holiday_router_arn]
}
statement {
sid = "SendHolidayJobs"
effect = "Allow"

View file

@ -2,16 +2,7 @@ resource "aws_ssm_parameter" "deploy_artifacts_bucket" {
name = "${local.ssm_prefix}/deploy/artifacts-bucket"
type = "String"
value = aws_s3_bucket.artifacts.id
description = "Lambda artifacts bucket; deploy.yaml uploads functions/<name>/<sha>.zip"
}
resource "aws_ssm_parameter" "deploy_function_name" {
for_each = local.functions
name = "${local.ssm_prefix}/deploy/${each.key}-function-name"
type = "String"
value = each.value.function_name
description = "Lambda function name for ${each.key}; deploy.yaml calls update-function-code"
description = "Unused leftover Lambda artifacts bucket from the dual-run zip path."
}
resource "aws_ssm_parameter" "deploy_api_url" {

View file

@ -1,4 +1,4 @@
"""Contracts for the HCP Terraform seam (PLAT-74)."""
"""Contracts for the HCP Terraform seam (PLAT-74 / PLAT-216)."""
from pathlib import Path
@ -6,7 +6,6 @@ ROOT = Path(__file__).resolve().parents[2]
TERRAFORM = ROOT / "terraform"
LAMBDA_TF = (TERRAFORM / "lambda.tf").read_text()
HCP_IAM = (TERRAFORM / "hcp_iam.tf").read_text()
DEPLOY = (ROOT / ".github" / "workflows" / "deploy.yaml").read_text()
CI = (ROOT / ".github" / "workflows" / "ci.yaml").read_text()
LOCALS = (TERRAFORM / "locals.tf").read_text()
VARIABLES = (TERRAFORM / "variables.tf").read_text()
@ -17,17 +16,11 @@ def test_sam_template_removed():
assert not (ROOT / "samconfig.toml.example").exists()
def test_lambda_ignore_changes_includes_code_attributes():
for attr in (
"filename",
"s3_bucket",
"s3_key",
"s3_object_version",
"source_code_hash",
):
assert attr in LAMBDA_TF
assert "lifecycle" in LAMBDA_TF
assert "ignore_changes" in LAMBDA_TF
def test_zip_cd_removed():
assert not (ROOT / ".github" / "workflows" / "deploy.yaml").exists()
assert 'resource "aws_lambda_function"' not in LAMBDA_TF
assert not (TERRAFORM / "apigateway.tf").exists()
assert not (TERRAFORM / "events.tf").exists()
def test_schedules_disabled_by_default():
@ -115,16 +108,6 @@ def test_in_repo_hcptf_roles():
assert "DenyCreatePolicy" in HCP_IAM
def test_deploy_workflow_is_prod_zip_cd():
assert "release: published" not in DEPLOY
assert "cd-sam" not in DEPLOY
assert "environment: prod" in DEPLOY
assert "deploy-afterhours-prod" in DEPLOY
assert "gh release create" not in DEPLOY
assert "package_lambdas.py" in DEPLOY
assert "update-function-code" in DEPLOY
def test_ci_runs_pytest_and_terraform_validate():
assert "ci-python-sam" not in CI
assert "pytest" in CI
@ -143,7 +126,7 @@ def test_checkcomponents_queue_arn_variable_matches_iam_references():
assert "checkcomponents_pair" in data_tf
def test_eight_functions_named():
def test_eight_function_names_still_on_leftover_roles():
for name in (
"afterhours-shift-manager",
"afterhours-weekly-post",
@ -161,14 +144,13 @@ def test_weekly_post_role_is_tf_managed_name():
assert 'role_name = "afterhours-shift-manager-weekly-post"' in LOCALS
def test_github_deploy_trust_covers_zip_and_image():
def test_github_deploy_trust_covers_image_only():
iam = (TERRAFORM / "iam_github_deploy.tf").read_text()
assert "environment:prod" in iam or "environment:prod" in LOCALS
assert "environment:dev" in iam or "environment:dev" in LOCALS
assert "deploy.yaml@refs/heads/${var.github_deploy_branch}" in iam
assert "deploy.yaml@" not in iam
assert "deploy-api.yaml@refs/heads/${var.github_deploy_branch}" in iam
assert "deploy-api.yaml@refs/tags/v*" in iam
assert "deploy.yaml@*" not in iam
assert "ecs:ListTasks" in iam
@ -176,3 +158,19 @@ def test_plan_refresh_includes_provider6_s3_gets():
assert "s3:GetLifecycleConfiguration" in HCP_IAM
assert "s3:GetReplicationConfiguration" in HCP_IAM
assert "s3:GetBucketReplication" in HCP_IAM
def test_origins_use_fargate_url():
outputs = (TERRAFORM / "outputs.tf").read_text()
assert "aws_apigatewayv2_api.http" not in outputs
assert '${local.api_url}/slack/events' in outputs
assert "value = local.api_url" in outputs
def test_alb_alarms_remain():
alarms = (TERRAFORM / "alarms.tf").read_text()
assert "ALB-5xx-" in alarms
assert "ALB-Latency-" in alarms
assert "ALB-UnhealthyHost-" in alarms
assert "ApiGateway-" not in alarms
assert "Lambda-" not in alarms