mirror of
https://github.com/Sea-Haven-Industries/afterhours-shift-manager.git
synced 2026-09-30 06:43:12 +00:00
Move HTTP and scheduled work onto one always-on Flask task so after-hours loses Lambda cold start without changing the Cognito or roster contracts.
205 lines
7.6 KiB
HCL
205 lines
7.6 KiB
HCL
locals {
|
|
lambda_alarm_matrix = {
|
|
errors = {
|
|
metric_name = "Errors"
|
|
statistic = "Sum"
|
|
evaluation_periods = 1
|
|
datapoints_to_alarm = 1
|
|
threshold = 1
|
|
comparison = "GreaterThanOrEqualToThreshold"
|
|
period = 300
|
|
}
|
|
throttles = {
|
|
metric_name = "Throttles"
|
|
statistic = "Sum"
|
|
evaluation_periods = 1
|
|
datapoints_to_alarm = 1
|
|
threshold = 1
|
|
comparison = "GreaterThanOrEqualToThreshold"
|
|
period = 300
|
|
}
|
|
}
|
|
|
|
lambda_alarms = {
|
|
for pair in flatten([
|
|
for fn_key, fn in local.functions : [
|
|
for metric_key, metric in local.lambda_alarm_matrix : {
|
|
key = "${fn_key}-${metric_key}"
|
|
fn_key = fn_key
|
|
function = fn.function_name
|
|
metric_key = metric_key
|
|
metric_name = metric.metric_name
|
|
statistic = metric.statistic
|
|
evaluation = metric.evaluation_periods
|
|
datapoints = metric.datapoints_to_alarm
|
|
threshold = metric.threshold
|
|
comparison = metric.comparison
|
|
period = metric.period
|
|
description = metric_key == "errors" ? "${fn.function_name} reported one or more errors" : "${fn.function_name} was throttled (concurrency limit hit)"
|
|
}
|
|
]
|
|
]) : pair.key => pair
|
|
}
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "lambda_errors_throttles" {
|
|
for_each = local.lambda_alarms
|
|
|
|
alarm_name = "Lambda-${title(each.value.metric_key)}-${each.value.function}"
|
|
alarm_description = each.value.description
|
|
namespace = "AWS/Lambda"
|
|
metric_name = each.value.metric_name
|
|
dimensions = { FunctionName = each.value.function }
|
|
statistic = each.value.statistic
|
|
period = each.value.period
|
|
evaluation_periods = each.value.evaluation
|
|
datapoints_to_alarm = each.value.datapoints
|
|
threshold = each.value.threshold
|
|
comparison_operator = each.value.comparison
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "lambda_duration" {
|
|
for_each = local.functions
|
|
|
|
alarm_name = "Lambda-Duration-${each.value.function_name}"
|
|
alarm_description = "${each.value.function_name} duration approaching its ${each.value.timeout}s timeout (>=${each.value.duration_ms}ms)"
|
|
namespace = "AWS/Lambda"
|
|
metric_name = "Duration"
|
|
dimensions = { FunctionName = each.value.function_name }
|
|
statistic = "Maximum"
|
|
period = 300
|
|
evaluation_periods = 3
|
|
datapoints_to_alarm = 2
|
|
threshold = each.value.duration_ms
|
|
comparison_operator = "GreaterThanOrEqualToThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "ddb_read_throttle" {
|
|
alarm_name = "DDB-ReadThrottle-${local.table_name}"
|
|
alarm_description = "afterhours-shifts table had one or more read throttle events"
|
|
namespace = "AWS/DynamoDB"
|
|
metric_name = "ReadThrottleEvents"
|
|
dimensions = { TableName = aws_dynamodb_table.shifts.name }
|
|
statistic = "Sum"
|
|
period = 300
|
|
evaluation_periods = 1
|
|
threshold = 0
|
|
comparison_operator = "GreaterThanThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "ddb_write_throttle" {
|
|
alarm_name = "DDB-WriteThrottle-${local.table_name}"
|
|
alarm_description = "afterhours-shifts table had one or more write throttle events"
|
|
namespace = "AWS/DynamoDB"
|
|
metric_name = "WriteThrottleEvents"
|
|
dimensions = { TableName = aws_dynamodb_table.shifts.name }
|
|
statistic = "Sum"
|
|
period = 300
|
|
evaluation_periods = 1
|
|
threshold = 0
|
|
comparison_operator = "GreaterThanThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "api_4xx" {
|
|
alarm_name = "ApiGateway-4xx-${aws_apigatewayv2_api.http.id}"
|
|
alarm_description = "Elevated 4xx responses on the afterhours HTTP API"
|
|
namespace = "AWS/ApiGateway"
|
|
metric_name = "4xx"
|
|
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
|
|
statistic = "Sum"
|
|
period = 300
|
|
evaluation_periods = 1
|
|
threshold = 5
|
|
comparison_operator = "GreaterThanOrEqualToThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "api_5xx" {
|
|
alarm_name = "ApiGateway-5xx-${aws_apigatewayv2_api.http.id}"
|
|
alarm_description = "5xx responses on the afterhours HTTP API"
|
|
namespace = "AWS/ApiGateway"
|
|
metric_name = "5xx"
|
|
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
|
|
statistic = "Sum"
|
|
period = 300
|
|
evaluation_periods = 1
|
|
threshold = 1
|
|
comparison_operator = "GreaterThanOrEqualToThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "api_latency" {
|
|
alarm_name = "ApiGateway-Latency-${aws_apigatewayv2_api.http.id}"
|
|
alarm_description = "p99 latency on the afterhours HTTP API exceeded 3s"
|
|
namespace = "AWS/ApiGateway"
|
|
metric_name = "Latency"
|
|
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
|
|
extended_statistic = "p99"
|
|
period = 300
|
|
evaluation_periods = 3
|
|
datapoints_to_alarm = 2
|
|
threshold = 3000
|
|
comparison_operator = "GreaterThanOrEqualToThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "alb_5xx" {
|
|
alarm_name = "ALB-5xx-${local.project}"
|
|
alarm_description = "ALB 5xx from afterhours-shift-manager"
|
|
namespace = "AWS/ApplicationELB"
|
|
metric_name = "HTTPCode_Target_5XX_Count"
|
|
dimensions = { LoadBalancer = aws_lb.api.arn_suffix }
|
|
statistic = "Sum"
|
|
period = 300
|
|
evaluation_periods = 1
|
|
threshold = 0
|
|
comparison_operator = "GreaterThanThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "alb_latency" {
|
|
alarm_name = "ALB-Latency-${local.project}"
|
|
alarm_description = "p99 target response time on the afterhours ALB exceeded 3s"
|
|
namespace = "AWS/ApplicationELB"
|
|
metric_name = "TargetResponseTime"
|
|
dimensions = { LoadBalancer = aws_lb.api.arn_suffix }
|
|
extended_statistic = "p99"
|
|
period = 300
|
|
evaluation_periods = 3
|
|
datapoints_to_alarm = 2
|
|
threshold = 3
|
|
comparison_operator = "GreaterThanOrEqualToThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "alb_unhealthy_hosts" {
|
|
alarm_name = "ALB-UnhealthyHost-${local.project}"
|
|
alarm_description = "Unhealthy Fargate targets on the afterhours ALB"
|
|
namespace = "AWS/ApplicationELB"
|
|
metric_name = "UnHealthyHostCount"
|
|
dimensions = {
|
|
LoadBalancer = aws_lb.api.arn_suffix
|
|
TargetGroup = aws_lb_target_group.api.arn_suffix
|
|
}
|
|
statistic = "Maximum"
|
|
period = 60
|
|
evaluation_periods = 3
|
|
datapoints_to_alarm = 3
|
|
threshold = 0
|
|
comparison_operator = "GreaterThanThreshold"
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
}
|