meal-order-manager/terraform/alarms.tf
Adam Moussa f48a82c476
Some checks are pending
Deploy API / Resolve target (push) Waiting to run
Deploy API / Deploy API to (push) Blocked by required conditions
feat(api): serve meals on ECS Fargate instead of Lambda (PLAT-215) (#199)
* feat(api): serve meals on ECS Fargate instead of Lambda

Keep the Flask app always-on with in-process jobs so CloudFront no longer fronts a cold-start API Gateway.

* fix(jobs): run delayed close and reminder deliveries

Wall-clock skip windows dropped the only weekly SQS attempt when Scheduler already fired in Eastern time. Dev schedules stay disabled.

* fix(api): return JSON objects and stop logging job payloads

Flask now jsonify-s handler dicts so API responses are not HTML, and the worker logs only event and status.

* fix(ci): restore the reusable workflow so the required check is named ci / ci

Inlining the job reported `ci` instead of the org ruleset's `ci / ci`.

* fix(secrets): drop unused os import so ruff check passes

* style: apply ruff format so ci-python-app lint passes

* fix(infra): give meals its own VPC because prod has none

* chore(security): re-key ALB SG checkov suppression after vpc.tf
2026-09-21 19:34:24 +00:00

102 lines
3.1 KiB
HCL

# CloudWatch alarms for the Fargate API and orders table.
resource "aws_cloudwatch_metric_alarm" "alb_5xx" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-alb-5xx"
alarm_description = "ALB 5xx from meal-order-manager"
namespace = "AWS/ApplicationELB"
metric_name = "HTTPCode_Target_5XX_Count"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
LoadBalancer = aws_lb.api.arn_suffix
}
}
resource "aws_cloudwatch_metric_alarm" "ecs_cpu" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-ecs-cpu"
alarm_description = "meal-order-manager ECS CPU above 80 percent"
namespace = "AWS/ECS"
metric_name = "CPUUtilization"
statistic = "Average"
period = 300
evaluation_periods = 2
threshold = 80
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
ClusterName = aws_ecs_cluster.api.name
ServiceName = aws_ecs_service.api.name
}
}
resource "aws_cloudwatch_metric_alarm" "jobs_dlq" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-jobs-dlq"
alarm_description = "Jobs DLQ is not empty"
namespace = "AWS/SQS"
metric_name = "ApproximateNumberOfMessagesVisible"
statistic = "Maximum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
QueueName = aws_sqs_queue.jobs_dlq.name
}
}
resource "aws_cloudwatch_metric_alarm" "dynamodb_read_throttles" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-ddb-read-throttles"
alarm_description = "Orders table read throttles"
namespace = "AWS/DynamoDB"
metric_name = "ReadThrottleEvents"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
TableName = aws_dynamodb_table.orders.name
}
}
resource "aws_cloudwatch_metric_alarm" "dynamodb_write_throttles" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-ddb-write-throttles"
alarm_description = "Orders table write throttles"
namespace = "AWS/DynamoDB"
metric_name = "WriteThrottleEvents"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
TableName = aws_dynamodb_table.orders.name
}
}