forgejo/lib/constructs/backup-verification.ts
Adam Moussa a7536f0bd7
Fix daily flapping of backup-verification errors alarm (#23)
The forgejo-backup-verification Lambda runs once per day, so its Errors
metric has data for only one hour and is missing for the other ~23h.
The errors alarm used TreatMissingData=BREACHING, which treated those
23h of missing data as a breach and flipped the alarm OK->ALARM every
day around 11:01 UTC despite zero actual errors.

Changes (alarm-only, no instance changes):
- ErrorAlarm: TreatMissingData BREACHING -> NOT_BREACHING. No data now
  means "no errors = healthy" instead of a false breach.
- Add forgejo-backup-verification-not-running: Invocations Sum over a
  24h period, alarms when < 1 invocation. This is the real "the daily
  verification never ran" guard that the BREACHING setting was trying
  (incorrectly) to provide.

Both alarms keep the existing action wiring (no SNS/OK actions), per the
org convention of never notifying on recovery.
2026-06-05 14:34:45 -04:00

136 lines
4.9 KiB
TypeScript

import * as cdk from "aws-cdk-lib";
import * as cloudwatch from "aws-cdk-lib/aws-cloudwatch";
import * as events from "aws-cdk-lib/aws-events";
import * as events_targets from "aws-cdk-lib/aws-events-targets";
import * as iam from "aws-cdk-lib/aws-iam";
import * as lambda from "aws-cdk-lib/aws-lambda";
import * as logs from "aws-cdk-lib/aws-logs";
import * as s3 from "aws-cdk-lib/aws-s3";
import { PythonFunction } from "@aws-cdk/aws-lambda-python-alpha";
import { Construct } from "constructs";
interface BackupVerificationProps {
sourceBucket: s3.IBucket;
replicaBucketName: string;
gcsBucket: string;
gcsSaSecretName: string;
slackWebhookSecretName: string;
}
export class BackupVerification extends Construct {
public readonly functionArn: string;
constructor(scope: Construct, id: string, props: BackupVerificationProps) {
super(scope, id);
const fn = new PythonFunction(this, "Function", {
functionName: "forgejo-backup-verification",
entry: "lambda/backup-verification",
runtime: lambda.Runtime.PYTHON_3_12,
architecture: lambda.Architecture.ARM_64,
handler: "handler",
index: "app.py",
memorySize: 512,
ephemeralStorageSize: cdk.Size.gibibytes(4),
timeout: cdk.Duration.minutes(5),
environment: {
SOURCE_BUCKET: props.sourceBucket.bucketName,
REPLICA_BUCKET: props.replicaBucketName,
GCS_BUCKET: props.gcsBucket,
GCS_SA_SECRET_NAME: props.gcsSaSecretName,
SLACK_WEBHOOK_SECRET_NAME: props.slackWebhookSecretName,
},
logRetention: logs.RetentionDays.TWO_MONTHS,
});
props.sourceBucket.grantRead(fn);
fn.addToRolePolicy(
new iam.PolicyStatement({
actions: ["s3:ListBucket", "s3:GetObject"],
resources: [
`arn:aws:s3:::${props.replicaBucketName}`,
`arn:aws:s3:::${props.replicaBucketName}/*`,
],
})
);
const account = cdk.Stack.of(this).account;
const region = cdk.Stack.of(this).region;
fn.addToRolePolicy(
new iam.PolicyStatement({
actions: ["secretsmanager:GetSecretValue"],
resources: [
`arn:aws:secretsmanager:${region}:${account}:secret:${props.gcsSaSecretName}-*`,
`arn:aws:secretsmanager:${region}:${account}:secret:${props.slackWebhookSecretName}-*`,
],
})
);
fn.addToRolePolicy(
new iam.PolicyStatement({
actions: ["ec2:DescribeSnapshots"],
resources: ["*"],
})
);
new events.Rule(this, "DailyCheck", {
ruleName: "forgejo-backup-daily-check",
schedule: events.Schedule.cron({ hour: "8", minute: "0" }),
targets: [
new events_targets.LambdaFunction(fn, {
event: events.RuleTargetInput.fromObject({ mode: "daily" }),
}),
],
});
new events.Rule(this, "MonthlyRestoreTest", {
ruleName: "forgejo-backup-monthly-restore-test",
schedule: events.Schedule.cron({
hour: "9",
minute: "0",
day: "1",
}),
targets: [
new events_targets.LambdaFunction(fn, {
event: events.RuleTargetInput.fromObject({ mode: "restore-test" }),
}),
],
});
// Errors alarm: only fires when the function actually runs and errors.
// The function runs once daily, so for ~23h there is no data. Treating
// missing data as BREACHING flipped this alarm OK->ALARM every day around
// 11:01 UTC even though no error ever occurred. NOT_BREACHING means "no
// data = no errors = healthy"; the separate not-running alarm below covers
// the "verification never ran" case.
new cloudwatch.Alarm(this, "ErrorAlarm", {
alarmName: "forgejo-backup-verification-errors",
alarmDescription: "Backup verification Lambda is failing — Slack notifications may not be firing",
metric: fn.metricErrors({ period: cdk.Duration.hours(1) }),
threshold: 1,
evaluationPeriods: 1,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
});
// Not-running alarm: fires if the daily verification did not invoke at all
// in a 24h window. This is the real "missing run" guard that the errors
// alarm's BREACHING setting was previously (and incorrectly) providing.
new cloudwatch.Alarm(this, "NotRunningAlarm", {
alarmName: "forgejo-backup-verification-not-running",
alarmDescription: "Backup verification Lambda has not run in the last 24h — daily verification may be broken",
metric: fn.metricInvocations({
period: cdk.Duration.hours(24),
statistic: cloudwatch.Stats.SUM,
}),
threshold: 1,
evaluationPeriods: 1,
treatMissingData: cloudwatch.TreatMissingData.BREACHING,
comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD,
});
this.functionArn = fn.functionArn;
}
}