Compare commits

...

7 commits

Author SHA1 Message Date
Adam Moussa
7aaf8340a5 Add 4 GiB ephemeral storage to verification Lambda
Monthly restore-test downloads and extracts the full dump tarball
in /tmp. As the dump grows with LFS data, the default 512 MB will
eventually cause ENOSPC failures.
2026-05-14 15:03:36 -04:00
Adam Moussa
c9936b1d2b Rename GCS service account to match read-only permissions 2026-05-14 14:35:45 -04:00
Adam Moussa
7ebea84abd Fix restore runbook: trailing-dot cp idiom and Glacier restore step 2026-05-14 14:22:27 -04:00
Adam Moussa
5cae537ee2 Fix restore runbook, DLM snapshot tagging, README cleanup, and gsutil prompt 2026-05-14 13:56:36 -04:00
Adam Moussa
bafb924cfc Fix EBS snapshot state check, drop unused GCS write grant and dead lifecycle rule 2026-05-14 13:24:24 -04:00
Adam Moussa
c771ed4f08 Rename SECRET_ARN env vars to SECRET_NAME to match actual values 2026-05-14 11:38:31 -04:00
Adam Moussa
d59623599c Fix GCS backup check: align staleness cutoff and add size validation
GCS check used a 72h cutoff but only listed 2 days of prefixes (~48h),
making the staleness check unreachable. Also added 1MB minimum file
size validation to match the S3 check.
2026-05-14 10:25:05 -04:00
6 changed files with 40 additions and 34 deletions

View file

@ -8,7 +8,7 @@ Self-hosted Forgejo git server for archiving GitHub repos and mirroring active o
- **Network**: Private subnet (us-east-1a), behind `seahaven-com` ALB for SSL termination
- **DNS**: `forgejo.seahaven.com` — Route53 alias record pointing to the `seahaven-com` ALB (not a direct A record)
- **TLS**: Wildcard cert on ALB, HTTP internally on port 3000
- **Backup**: Nightly `forgejo dump` to S3 + EBS snapshots via DLM (see [S3 Backups](#s3-backups))
- **Backup**: Nightly `forgejo dump` to S3 + EBS snapshots via DLM (see [3-2-1 Backup Strategy](#3-2-1-backup-strategy))
- **Admin access**: SSM Session Manager (no SSH port exposed)
- **CI/CD**: GitHub Actions with OIDC role `githubdeploy-forgejo`
@ -65,6 +65,17 @@ sudo /usr/local/bin/forgejo-backup.sh
### Restore from S3
For backups older than 30 days (Glacier), restore the object first:
```bash
aws s3api restore-object --bucket forgejo-backups-328440206208 \
--key "archive/<date>/forgejo-<date>.tar.gz" \
--restore-request '{"Days":7,"GlacierJobParameters":{"Tier":"Standard"}}'
# Wait ~3-5 hours for restore to complete, then:
```
Download and restore:
```bash
aws s3 cp s3://forgejo-backups-328440206208/archive/<date>/forgejo-<date>.tar.gz /tmp/
systemctl stop forgejo
@ -74,6 +85,9 @@ cp app.ini /etc/forgejo/app.ini
cp gitea-db.sqlite3 /var/lib/forgejo/data/forgejo.db
rm -rf /var/lib/forgejo/data/repositories
cp -a repos /var/lib/forgejo/data/repositories
cp -a data/. /var/lib/forgejo/data/
[ -d lfs ] && cp -a lfs/. /var/lib/forgejo/data/lfs/
[ -d custom ] && cp -a custom/. /var/lib/forgejo/custom/
chown -R forgejo:forgejo /var/lib/forgejo /etc/forgejo/app.ini
systemctl start forgejo
rm -rf /tmp/forgejo-restore /tmp/forgejo-<date>.tar.gz
@ -87,12 +101,6 @@ gsutil cp gs://forgejo-backups-offsite-seahaven/archive/<date>/forgejo-<date>.ta
# Then follow the same restore steps as S3 above
```
To test the backup manually:
```bash
sudo /usr/local/bin/forgejo-backup.sh
```
## Autodiscovery
An hourly cron job checks the `Sea-Haven-Industries` GitHub org for new repositories and mirrors them into Forgejo automatically.

View file

@ -18,8 +18,8 @@ secrets = boto3.client("secretsmanager")
SOURCE_BUCKET = os.environ["SOURCE_BUCKET"]
REPLICA_BUCKET = os.environ["REPLICA_BUCKET"]
GCS_BUCKET = os.environ["GCS_BUCKET"]
GCS_SA_SECRET_ARN = os.environ["GCS_SA_SECRET_ARN"]
SLACK_WEBHOOK_SECRET_ARN = os.environ["SLACK_WEBHOOK_SECRET_ARN"]
GCS_SA_SECRET_NAME = os.environ["GCS_SA_SECRET_NAME"]
SLACK_WEBHOOK_SECRET_NAME = os.environ["SLACK_WEBHOOK_SECRET_NAME"]
_gcs_client = None
@ -27,7 +27,7 @@ _gcs_client = None
def _get_gcs_client():
global _gcs_client
if _gcs_client is None:
raw = secrets.get_secret_value(SecretId=GCS_SA_SECRET_ARN)["SecretString"]
raw = secrets.get_secret_value(SecretId=GCS_SA_SECRET_NAME)["SecretString"]
info = json.loads(raw)
creds = service_account.Credentials.from_service_account_info(info)
_gcs_client = gcs.Client(credentials=creds, project=info.get("project_id"))
@ -69,11 +69,13 @@ def _check_gcs():
blobs.extend(list(bucket.list_blobs(prefix=f"archive/{date_prefix}/")))
if not blobs:
return False, "GCS Offsite: No objects found under archive/ for last 2 days"
cutoff = now - timedelta(hours=72)
cutoff = now - timedelta(hours=48)
latest = max(blobs, key=lambda b: b.updated)
if latest.updated < cutoff:
age = (now - latest.updated).total_seconds() / 3600
return False, f"GCS Offsite: Latest object is {age:.0f}h old ({latest.name})"
if latest.size < 1_000_000:
return False, f"GCS Offsite: Latest dump suspiciously small ({latest.size} bytes)"
return True, f"GCS Offsite: OK — {latest.name} ({latest.size / 1_000_000:.1f} MB)"
except Exception as e:
return False, f"GCS Offsite: Error — {e}"
@ -90,12 +92,18 @@ def _check_ebs_snapshots():
snapshots = resp.get("Snapshots", [])
if not snapshots:
return False, "EBS Snapshots: No snapshots found with forgejo-backup tag"
recent = [s for s in snapshots if s["StartTime"] >= cutoff]
recent = [s for s in snapshots if s["StartTime"] >= cutoff and s.get("State") == "completed"]
if not recent:
pending = sum(1 for s in snapshots if s["StartTime"] >= cutoff and s.get("State") == "pending")
errored = sum(1 for s in snapshots if s["StartTime"] >= cutoff and s.get("State") == "error")
latest = max(snapshots, key=lambda s: s["StartTime"])
age = (now - latest["StartTime"]).total_seconds() / 3600
return False, f"EBS Snapshots: Latest is {age:.0f}h old ({latest['SnapshotId']})"
return True, f"EBS Snapshots: OK — {len(snapshots)} total, {len(recent)} in last 48h"
return False, (
f"EBS Snapshots: No completed snapshot in last 48h "
f"(latest {age:.0f}h old, state={latest.get('State')}; "
f"pending={pending}, error={errored})"
)
return True, f"EBS Snapshots: OK — {len(recent)} completed in last 48h"
except Exception as e:
return False, f"EBS Snapshots: Error — {e}"
@ -165,7 +173,7 @@ def _restore_test():
def _post_slack(blocks):
raw = secrets.get_secret_value(SecretId=SLACK_WEBHOOK_SECRET_ARN)["SecretString"]
raw = secrets.get_secret_value(SecretId=SLACK_WEBHOOK_SECRET_NAME)["SecretString"]
webhook_url = raw.strip()
payload = json.dumps({"blocks": blocks}).encode()
req = urllib.request.Request(

View file

@ -28,13 +28,14 @@ export class BackupVerification extends Construct {
handler: "handler",
index: "app.py",
memorySize: 512,
ephemeralStorageSize: cdk.Size.gibibytes(4),
timeout: cdk.Duration.minutes(5),
environment: {
SOURCE_BUCKET: props.sourceBucket.bucketName,
REPLICA_BUCKET: props.replicaBucketName,
GCS_BUCKET: props.gcsBucket,
GCS_SA_SECRET_ARN: props.gcsSaSecretName,
SLACK_WEBHOOK_SECRET_ARN: props.slackWebhookSecretName,
GCS_SA_SECRET_NAME: props.gcsSaSecretName,
SLACK_WEBHOOK_SECRET_NAME: props.slackWebhookSecretName,
},
logRetention: logs.RetentionDays.TWO_MONTHS,
});

View file

@ -26,17 +26,6 @@ export class ForgejoReplicaStack extends cdk.Stack {
},
],
},
{
id: "mirror-to-glacier-then-expire",
prefix: "mirror/",
transitions: [
{
storageClass: s3.StorageClass.GLACIER,
transitionAfter: cdk.Duration.days(30),
},
],
expiration: cdk.Duration.days(365),
},
{
id: "cleanup-noncurrent-versions",
noncurrentVersionExpiration: cdk.Duration.days(90),

View file

@ -381,6 +381,7 @@ export class ForgejoStack extends cdk.Stack {
createRule: { interval: 24, intervalUnit: "HOURS", times: ["06:00"] },
retainRule: { count: 30 },
copyTags: true,
tagsToAdd: [{ key: "forgejo-backup", value: "true" }],
}],
},
});

View file

@ -4,7 +4,7 @@ set -euo pipefail
PROJECT_ID="sea-haven-backups"
BUCKET_NAME="forgejo-backups-offsite-seahaven"
LOCATION="us-central1"
SA_NAME="forgejo-backup-writer"
SA_NAME="forgejo-backup-verifier"
SA_EMAIL="${SA_NAME}@${PROJECT_ID}.iam.gserviceaccount.com"
RETENTION_SECONDS=$((2 * 365 * 24 * 3600)) # 2 years
AWS_REGION="us-east-1"
@ -76,7 +76,7 @@ echo "Even the project owner cannot shorten or remove the policy."
echo ""
read -p "Lock the retention policy now? (yes/no): " CONFIRM
if [ "$CONFIRM" = "yes" ]; then
$GSUTIL retention lock "gs://$BUCKET_NAME"
echo y | $GSUTIL retention lock "gs://$BUCKET_NAME"
echo "Retention policy LOCKED."
else
echo "Retention policy set but NOT locked. Run 'gsutil retention lock gs://$BUCKET_NAME' when ready."
@ -89,16 +89,15 @@ if $GCLOUD iam service-accounts describe "$SA_EMAIL" &>/dev/null 2>&1; then
echo "Service account $SA_EMAIL already exists."
else
$GCLOUD iam service-accounts create "$SA_NAME" \
--display-name="Forgejo Backup Writer" \
--description="Write-only access to forgejo offsite backup bucket"
--display-name="Forgejo Backup Verifier" \
--description="Read-only access to forgejo offsite backup bucket (verification Lambda)"
echo "Created service account $SA_EMAIL."
fi
echo ""
echo "--- Step 8: Grant bucket permissions ---"
$GSUTIL iam ch "serviceAccount:${SA_EMAIL}:objectCreator" "gs://$BUCKET_NAME"
$GSUTIL iam ch "serviceAccount:${SA_EMAIL}:objectViewer" "gs://$BUCKET_NAME"
echo "Granted objectCreator + objectViewer to $SA_EMAIL."
echo "Granted objectViewer to $SA_EMAIL."
echo ""
echo "--- Step 9: Create and store service account key ---"