forgejo/terraform/user_data.sh

259 lines
9 KiB
Bash
Raw Normal View History

#!/bin/bash
set -euxo pipefail
FORGEJO_VERSION="__FORGEJO_VERSION__"
BACKUP_BUCKET="__BACKUP_BUCKET__"
AWS_REGION="__AWS_REGION__"
dnf install -y git cronie python3
systemctl enable --now crond
useradd --system --shell /bin/bash --home-dir /home/forgejo --create-home forgejo
# Persistent data volume. Wait until the attachment shows up, and never format a disk that already has a filesystem.
root_disk() {
lsblk -no PKNAME "$(findmnt -n -o SOURCE /)" 2>/dev/null || echo xvda
}
data_disk() {
local root name
root="$(root_disk)"
while read -r name; do
if [ -n "$name" ] && [ "$name" != "$root" ]; then
printf '%s\n' "$name"
return 0
fi
done < <(lsblk -dno NAME)
return 1
}
until data_disk >/dev/null; do
echo "Waiting for data volume..."
sleep 5
done
DATA_DEVICE="/dev/$(data_disk)"
if ! blkid "$DATA_DEVICE"; then
mkfs.ext4 -L forgejo-data "$DATA_DEVICE"
fi
mkdir -p /var/lib/forgejo
echo "LABEL=forgejo-data /var/lib/forgejo ext4 defaults,nofail 0 2" >> /etc/fstab
mount -a
mkdir -p /var/lib/forgejo/{data,log}
chown -R forgejo:forgejo /var/lib/forgejo
chmod 750 /var/lib/forgejo
curl -Lo /usr/local/bin/forgejo "https://codeberg.org/forgejo/forgejo/releases/download/v${FORGEJO_VERSION}/forgejo-${FORGEJO_VERSION}-linux-arm64"
chmod +x /usr/local/bin/forgejo
mkdir -p /etc/forgejo
chown root:forgejo /etc/forgejo
chmod 770 /etc/forgejo
cat > /etc/forgejo/app.ini << 'INIEOF'
APP_NAME = Sea Haven Git
[server]
DOMAIN = forgejo.seahaven.com
ROOT_URL = https://forgejo.seahaven.com/
HTTP_PORT = 3000
START_SSH_SERVER = true
SSH_PORT = 2222
SSH_LISTEN_PORT = 2222
LFS_START_SERVER = true
[database]
DB_TYPE = sqlite3
PATH = /var/lib/forgejo/data/forgejo.db
[repository]
ROOT = /var/lib/forgejo/data/repositories
[log]
ROOT_PATH = /var/lib/forgejo/log
[security]
INSTALL_LOCK = true
[service]
DISABLE_REGISTRATION = true
[mirror]
DEFAULT_INTERVAL = 1h
INIEOF
chown root:forgejo /etc/forgejo/app.ini
chmod 660 /etc/forgejo/app.ini
cat > /etc/systemd/system/forgejo.service << 'SVCEOF'
[Unit]
Description=Forgejo
After=network.target
[Service]
Type=simple
User=forgejo
Group=forgejo
WorkingDirectory=/var/lib/forgejo
ExecStart=/usr/local/bin/forgejo web --config /etc/forgejo/app.ini
Restart=always
RestartSec=5
Environment=USER=forgejo HOME=/home/forgejo FORGEJO_WORK_DIR=/var/lib/forgejo
[Install]
WantedBy=multi-user.target
SVCEOF
systemctl daemon-reload
# Restore from the latest S3 dump when the data volume has no database.
if [ ! -f /var/lib/forgejo/data/forgejo.db ]; then
echo "No database on data volume - restoring latest backup from S3"
S3_PREFIX="$(aws ssm get-parameter --name /forgejo/backup-s3-prefix --query Parameter.Value --output text --region "${AWS_REGION}" || echo archive)"
LATEST="$(aws s3 ls "s3://${BACKUP_BUCKET}/${S3_PREFIX}/" --region "${AWS_REGION}" | awk '{print $2}' | sort | tail -1 | tr -d '/')"
if [ -n "$LATEST" ]; then
FILE="$(aws s3 ls "s3://${BACKUP_BUCKET}/${S3_PREFIX}/${LATEST}/" --region "${AWS_REGION}" | awk '{print $4}' | tail -1)"
fix(terraform): allow HCP refresh of the log group and parameter (PLAT-80) (#101) * fix(terraform): allow HCP refresh of the log group and parameter The scoped apply role can create those resources, but CloudWatch and SSM list them on a wildcard ARN. DescribeLogGroups and DescribeParameters need that resource. * fix(terraform): let the plan role read bucket website config The S3 provider refreshes GetBucketWebsite. The plan role was denied on the three Forgejo buckets. * fix(terraform): let the plan role read backup object metadata HeadObject on the Lambda zip is s3:GetObject. The plan role only had the bucket ARNs. * fix(terraform): let the plan role read object tags and retention The S3 provider refreshes tagging, ACL, attributes, and Object Lock on the Lambda zip. * fix(terraform): scope plan object reads and restore on the data volume The plan role only needs object reads for the verification zip. Unpacking a dump in /tmp fills the root volume. * fix(terraform): accept dumps that already contain data/forgejo.db Today's archive has no gitea-db.sqlite3 at the root. Copy that file only when it is present. * docs(terraform): keep restore runbook on one bucket and fail closed Glacier and the download use the prod bucket. GCS unpacks on the data volume. Neither path deletes live repos until the dump has a database. * docs(terraform): keep optional restore copies from aborting under set -e if/fi matches user_data.sh. The file comment now says forgejo-services versions also go through the org-account role. * fix(terraform): restore the dump app.ini with the database INTERNAL_TOKEN, JWT_SECRET, and LFS_JWT_SECRET live in that file. A restore that keeps the generated file cannot decrypt the dumped secrets.
2026-09-30 00:10:09 +00:00
# The root volume is 20 GB. A dump expands well past that, so unpack on the data volume.
RESTORE_DIR=/var/lib/forgejo/.restore
rm -rf "$RESTORE_DIR"
mkdir -p "$RESTORE_DIR"
aws s3 cp "s3://${BACKUP_BUCKET}/${S3_PREFIX}/${LATEST}/${FILE}" - --region "${AWS_REGION}" --no-progress | tar -xz -C "$RESTORE_DIR"
mkdir -p /var/lib/forgejo/data /var/lib/forgejo/custom
if [ -d "$RESTORE_DIR/data" ]; then
cp -a "$RESTORE_DIR"/data/. /var/lib/forgejo/data/
fi
rm -rf /var/lib/forgejo/data/repositories
mkdir -p /var/lib/forgejo/data/repositories
if [ -d "$RESTORE_DIR/repos" ]; then
cp -a "$RESTORE_DIR"/repos/. /var/lib/forgejo/data/repositories/
fi
fix(terraform): allow HCP refresh of the log group and parameter (PLAT-80) (#101) * fix(terraform): allow HCP refresh of the log group and parameter The scoped apply role can create those resources, but CloudWatch and SSM list them on a wildcard ARN. DescribeLogGroups and DescribeParameters need that resource. * fix(terraform): let the plan role read bucket website config The S3 provider refreshes GetBucketWebsite. The plan role was denied on the three Forgejo buckets. * fix(terraform): let the plan role read backup object metadata HeadObject on the Lambda zip is s3:GetObject. The plan role only had the bucket ARNs. * fix(terraform): let the plan role read object tags and retention The S3 provider refreshes tagging, ACL, attributes, and Object Lock on the Lambda zip. * fix(terraform): scope plan object reads and restore on the data volume The plan role only needs object reads for the verification zip. Unpacking a dump in /tmp fills the root volume. * fix(terraform): accept dumps that already contain data/forgejo.db Today's archive has no gitea-db.sqlite3 at the root. Copy that file only when it is present. * docs(terraform): keep restore runbook on one bucket and fail closed Glacier and the download use the prod bucket. GCS unpacks on the data volume. Neither path deletes live repos until the dump has a database. * docs(terraform): keep optional restore copies from aborting under set -e if/fi matches user_data.sh. The file comment now says forgejo-services versions also go through the org-account role. * fix(terraform): restore the dump app.ini with the database INTERNAL_TOKEN, JWT_SECRET, and LFS_JWT_SECRET live in that file. A restore that keeps the generated file cannot decrypt the dumped secrets.
2026-09-30 00:10:09 +00:00
# Older dumps keep sqlite at the archive root. Current dumps already have data/forgejo.db.
if [ -f "$RESTORE_DIR/gitea-db.sqlite3" ]; then
cp "$RESTORE_DIR/gitea-db.sqlite3" /var/lib/forgejo/data/forgejo.db
fi
if [ -d "$RESTORE_DIR/lfs" ]; then
mkdir -p /var/lib/forgejo/data/lfs
cp -a "$RESTORE_DIR"/lfs/. /var/lib/forgejo/data/lfs/
fi
if [ -d "$RESTORE_DIR/custom" ]; then
cp -a "$RESTORE_DIR"/custom/. /var/lib/forgejo/custom/
fi
fix(terraform): allow HCP refresh of the log group and parameter (PLAT-80) (#101) * fix(terraform): allow HCP refresh of the log group and parameter The scoped apply role can create those resources, but CloudWatch and SSM list them on a wildcard ARN. DescribeLogGroups and DescribeParameters need that resource. * fix(terraform): let the plan role read bucket website config The S3 provider refreshes GetBucketWebsite. The plan role was denied on the three Forgejo buckets. * fix(terraform): let the plan role read backup object metadata HeadObject on the Lambda zip is s3:GetObject. The plan role only had the bucket ARNs. * fix(terraform): let the plan role read object tags and retention The S3 provider refreshes tagging, ACL, attributes, and Object Lock on the Lambda zip. * fix(terraform): scope plan object reads and restore on the data volume The plan role only needs object reads for the verification zip. Unpacking a dump in /tmp fills the root volume. * fix(terraform): accept dumps that already contain data/forgejo.db Today's archive has no gitea-db.sqlite3 at the root. Copy that file only when it is present. * docs(terraform): keep restore runbook on one bucket and fail closed Glacier and the download use the prod bucket. GCS unpacks on the data volume. Neither path deletes live repos until the dump has a database. * docs(terraform): keep optional restore copies from aborting under set -e if/fi matches user_data.sh. The file comment now says forgejo-services versions also go through the org-account role. * fix(terraform): restore the dump app.ini with the database INTERNAL_TOKEN, JWT_SECRET, and LFS_JWT_SECRET live in that file. A restore that keeps the generated file cannot decrypt the dumped secrets.
2026-09-30 00:10:09 +00:00
# The dump's app.ini carries INTERNAL_TOKEN, JWT_SECRET, and LFS_JWT_SECRET.
if [ -f "$RESTORE_DIR/app.ini" ]; then
cp "$RESTORE_DIR/app.ini" /etc/forgejo/app.ini
chown root:forgejo /etc/forgejo/app.ini
chmod 660 /etc/forgejo/app.ini
fi
chown -R forgejo:forgejo /var/lib/forgejo
rm -rf "$RESTORE_DIR"
if [ ! -s /var/lib/forgejo/data/forgejo.db ]; then
echo "Restore did not produce /var/lib/forgejo/data/forgejo.db" >&2
exit 1
fi
else
echo "No backup found in S3 - starting fresh"
fi
fi
systemctl enable --now forgejo
cat > /usr/local/bin/forgejo-backup.sh << 'BAKEOF'
#!/bin/bash
set -euo pipefail
TIMESTAMP="$(date +%Y-%m-%d)"
S3_PREFIX="$(aws ssm get-parameter --name /forgejo/backup-s3-prefix --query Parameter.Value --output text --region __AWS_REGION__ || echo archive)"
DUMP_DIR="$(mktemp -d)"
chown forgejo:forgejo "$DUMP_DIR"
cd "$DUMP_DIR"
sudo -u forgejo /usr/local/bin/forgejo dump --config /etc/forgejo/app.ini --type tar.gz --file "$DUMP_DIR/forgejo-${TIMESTAMP}.tar.gz"
aws s3 cp "$DUMP_DIR/forgejo-${TIMESTAMP}.tar.gz" "s3://__BACKUP_BUCKET__/${S3_PREFIX}/${TIMESTAMP}/forgejo-${TIMESTAMP}.tar.gz" --region __AWS_REGION__
rm -rf "$DUMP_DIR"
BAKEOF
chmod +x /usr/local/bin/forgejo-backup.sh
echo "0 5 * * * root /usr/local/bin/forgejo-backup.sh >> /var/log/forgejo-backup.log 2>&1" > /etc/cron.d/forgejo-backup
chmod 644 /etc/cron.d/forgejo-backup
cat > /usr/local/bin/forgejo-autodiscover.sh << 'ADEOF'
#!/bin/bash
set -euo pipefail
GH_PAT="$(aws secretsmanager get-secret-value --secret-id forgejo/github-pat --query SecretString --output text --region __AWS_REGION__)"
FORGEJO_TOKEN="$(aws secretsmanager get-secret-value --secret-id forgejo/api-token --query SecretString --output text --region __AWS_REGION__)"
FORGEJO_URL="https://forgejo.seahaven.com/api/v1"
GH_ORG="Sea-Haven-Industries"
gh_repos="$(curl -sf -H "Authorization: token ${GH_PAT}" "https://api.github.com/orgs/${GH_ORG}/repos?per_page=100&type=all" | python3 -c '
import json, sys
for r in json.load(sys.stdin):
print("%s\t%s" % (r["name"], r["archived"]))
')"
forgejo_repos="$(curl -sf -H "Authorization: token ${FORGEJO_TOKEN}" "${FORGEJO_URL}/repos/search?limit=100" | python3 -c '
import json, sys
data = json.load(sys.stdin)
repos = data.get("data", data) if isinstance(data, dict) else data
for r in repos:
print(r["name"])
')"
while IFS=$'\t' read -r name archived; do
if ! echo "$forgejo_repos" | grep -qx "$name"; then
mirror=true
[ "$archived" = "True" ] && mirror=false
echo "$(date -Is) Discovering: $name (mirror=$mirror)"
curl -sf -X POST "${FORGEJO_URL}/repos/migrate" \
-H "Authorization: token ${FORGEJO_TOKEN}" \
-H "Content-Type: application/json" \
-d "{
\"clone_addr\": \"https://github.com/${GH_ORG}/${name}.git\",
\"auth_token\": \"${GH_PAT}\",
\"repo_name\": \"${name}\",
\"repo_owner\": \"adam\",
\"service\": \"github\",
\"mirror\": ${mirror},
\"issues\": true,
\"labels\": true,
\"milestones\": true,
\"pull_requests\": true,
\"releases\": true,
\"wiki\": true
}" > /dev/null
fi
done <<< "$gh_repos"
ADEOF
chmod +x /usr/local/bin/forgejo-autodiscover.sh
cat > /usr/local/bin/forgejo-refresh-tokens.sh << 'RTEOF'
#!/bin/bash
set -euo pipefail
GH_PAT="$(aws secretsmanager get-secret-value --secret-id forgejo/github-pat --query SecretString --output text --region __AWS_REGION__)"
FORGEJO_TOKEN="$(aws secretsmanager get-secret-value --secret-id forgejo/api-token --query SecretString --output text --region __AWS_REGION__)"
FORGEJO_URL="https://forgejo.seahaven.com/api/v1"
REPO_ROOT="/var/lib/forgejo/data/repositories/adam"
export GIT_CONFIG_COUNT=1
export GIT_CONFIG_KEY_0=safe.directory
export GIT_CONFIG_VALUE_0='*'
mirrors="$(curl -sf -H "Authorization: token ${FORGEJO_TOKEN}" "${FORGEJO_URL}/repos/search?limit=100" | python3 -c '
import json, sys
data = json.load(sys.stdin)
repos = data.get("data", data) if isinstance(data, dict) else data
for r in repos:
if r.get("mirror", False):
print(r["name"])
')"
while read -r repo_name; do
[ -z "$repo_name" ] && continue
repo_dir="${REPO_ROOT}/${repo_name}.git"
if [ -d "$repo_dir" ]; then
new_url="https://${GH_PAT}@github.com/Sea-Haven-Industries/${repo_name}.git"
git -C "$repo_dir" remote set-url origin "$new_url" 2>/dev/null && echo "$(date -Is) Refreshed: ${repo_name}"
fi
done <<< "$mirrors"
RTEOF
chmod +x /usr/local/bin/forgejo-refresh-tokens.sh
printf '%s\n' '0 * * * * root /usr/local/bin/forgejo-autodiscover.sh >> /var/log/forgejo-autodiscover.log 2>&1' > /etc/cron.d/forgejo-autodiscover
printf '%s\n' '30 4 * * * root /usr/local/bin/forgejo-refresh-tokens.sh >> /var/log/forgejo-refresh-tokens.log 2>&1' > /etc/cron.d/forgejo-refresh-tokens
chmod 644 /etc/cron.d/forgejo-autodiscover /etc/cron.d/forgejo-refresh-tokens