.github/.github/workflows/callable-select-runner.yaml

278 lines
12 KiB
YAML

name: Select runner
# Picks the `runs-on` label for a caller's Linux jobs: GitHub-hosted while
# GitHub Actions is operational, the org self-hosted runner when it is not.
#
# How it decides, in order:
# 1. Fork pull requests always get `primary`. Fork code must never execute on
# a machine the org owns.
# 2. If the caller repo (or org) defines the Actions variable
# CI_RUNNER_OVERRIDE, its value is used verbatim. Set it to `self-hosted`
# to exercise the fallback path on demand, or to `ubuntu-latest` to pin
# hosted runners if the status page misreports. Delete it to return to
# automatic selection.
# 3. The public GitHub status page is consulted. `fallback` is returned
# when the Actions component reports `partial_outage`, `major_outage`,
# or `under_maintenance`. `degraded_performance` and an unreadable
# status page do not trigger fallback on their own. Degraded means
# slow-but-working, and one self-hosted box serialising every run is
# slower than that; treating "unknown" as healthy means a network
# problem on the self-hosted box does not silently route every run
# onto it.
# 4. The job queue is consulted. If any job targeting `primary` has sat in
# `queued` for more than `max-queue-minutes` (default 5), hosted runners
# are not picking up work regardless of what the status page says, and
# `fallback` is returned. This catches the common failure mode of
# Actions "operational" on paper but queueing in practice.
#
# Scope depends on credentials. With the org `sea-haven-runner-selector`
# GitHub App (secrets RUNNER_SELECTOR_APP_ID and
# RUNNER_SELECTOR_APP_PRIVATE_KEY, passed via `secrets: inherit`) the
# scan covers every non-archived org repo pushed in the last 24h, up to
# `scan-repo-limit`. Without the app it covers only the calling repo
# using GITHUB_TOKEN, and the caller must grant `actions: read`. If
# nothing can be read the check logs a warning and is skipped. Either
# way it only observes jobs that are already queued somewhere; a quiet
# org gets no signal here and relies on step 3.
#
# What this cannot do: move a job that is already queued on a hosted runner.
# `timeout-minutes` does not start until a runner picks the job up, and a
# job's `runs-on` is fixed once created. Fallback is decided up front, once
# per run, before any downstream job is queued.
#
# This job itself runs on `fallback`. That is the only runner that can be
# expected to pick up work when hosted runners are down, so every workflow
# that adopts this selector makes the self-hosted runner a hard dependency.
# One org runner exists today; concurrent runs serialise on it. Keep this job
# short.
#
# There is no native `runs-on` fallback in GitHub Actions. An array of labels
# is an AND match, not an ordered preference, and a job whose labels match no
# online runner sits queued for 24 hours before failing. This selector plus
# the `runner` input on every org reusable is the supported substitute.
#
# Caller example:
# jobs:
# select-runner:
# uses: Sea-Haven-Industries/.github/.github/workflows/callable-select-runner.yaml@<sha> # vX.Y.Z
# permissions:
# actions: read
# secrets: inherit
# ci:
# needs: select-runner
# uses: Sea-Haven-Industries/.github/.github/workflows/ci-terraform.yaml@<sha> # vX.Y.Z
# with:
# runner: ${{ needs.select-runner.outputs.runner }}
# own-job:
# needs: select-runner
# runs-on: ${{ needs.select-runner.outputs.runner || 'ubuntu-latest' }}
#
# The self-hosted runner needs curl and jq for this job, and whatever the
# downstream jobs need (python3, shellcheck, Node/Python tool-cache access,
# Chromium system libraries, and so on). Steps that use `sudo` on
# ubuntu-latest must branch on the selected runner.
on:
workflow_call:
inputs:
primary:
description: "Runner label returned while GitHub Actions is operational."
type: string
required: false
default: ubuntu-latest
fallback:
description: >-
Runner label returned when GitHub Actions is not operational. This
selector job runs on this label.
type: string
required: false
default: self-hosted
max-queue-minutes:
description: >-
Return `fallback` when any job targeting `primary` has been queued
longer than this many minutes. With the app secrets the whole org is
scanned; without them only the calling repo is, and the caller must
grant `actions: read`. 0 disables the queue check.
type: number
required: false
default: 5
scan-repo-limit:
description: >-
Org-wide scan only. Maximum number of non-archived repos to inspect,
most recently pushed first, and only those pushed in the last 24h.
type: number
required: false
default: 30
secrets:
RUNNER_SELECTOR_APP_ID:
description: >-
App ID of the org `sea-haven-runner-selector` GitHub App (Actions
read, Metadata read, Self-hosted runners read). Pass with
`secrets: inherit`. Optional; without it the queue check is limited
to the calling repo.
required: false
RUNNER_SELECTOR_APP_PRIVATE_KEY:
description: "Private key for RUNNER_SELECTOR_APP_ID. Optional."
required: false
outputs:
runner:
description: "Runner label for downstream `runs-on` and `runner` inputs."
value: ${{ jobs.select.outputs.runner }}
permissions:
actions: read
jobs:
select:
runs-on: ${{ inputs.fallback }}
timeout-minutes: 5
outputs:
runner: ${{ steps.pick.outputs.runner }}
env:
# `secrets` is not available in step-level `if`; surface presence here.
HAS_APP: ${{ secrets.RUNNER_SELECTOR_APP_ID != '' && secrets.RUNNER_SELECTOR_APP_PRIVATE_KEY != '' }}
steps:
- name: Mint org-wide read token
id: app-token
if: env.HAS_APP == 'true' && inputs.max-queue-minutes > 0
continue-on-error: true
uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0
with:
app-id: ${{ secrets.RUNNER_SELECTOR_APP_ID }}
private-key: ${{ secrets.RUNNER_SELECTOR_APP_PRIVATE_KEY }}
owner: ${{ github.repository_owner }}
- name: Pick runner
id: pick
env:
PRIMARY: ${{ inputs.primary }}
FALLBACK: ${{ inputs.fallback }}
MAX_QUEUE_MINUTES: ${{ inputs.max-queue-minutes }}
SCAN_REPO_LIMIT: ${{ inputs.scan-repo-limit }}
OVERRIDE: ${{ vars.CI_RUNNER_OVERRIDE }}
IS_FORK: ${{ github.event.pull_request.head.repo.fork }}
# App token when minted, else the repo-scoped GITHUB_TOKEN.
APP_TOKEN: ${{ steps.app-token.outputs.token }}
REPO_TOKEN: ${{ github.token }}
shell: bash
run: |
set -euo pipefail
pick() {
echo "Selected runner: $1 ($2)"
echo "runner=$1" >> "$GITHUB_OUTPUT"
}
if [ "$IS_FORK" = "true" ]; then
pick "$PRIMARY" "fork pull request"
exit 0
fi
if [ -n "$OVERRIDE" ]; then
pick "$OVERRIDE" "CI_RUNNER_OVERRIDE"
exit 0
fi
actions_status=$(
curl -sf --max-time 10 https://www.githubstatus.com/api/v2/components.json \
| jq -r '.components[] | select(.name == "Actions") | .status' \
|| true
)
actions_status=${actions_status:-unknown}
echo "GitHub Actions status: $actions_status"
# Statuspage component values: operational, degraded_performance,
# partial_outage, major_outage, under_maintenance.
case "$actions_status" in
partial_outage|major_outage|under_maintenance)
pick "$FALLBACK" "status $actions_status"
exit 0 ;;
esac
# Second signal: are hosted jobs already sitting in queue?
# Jobs blocked on `needs` are not listed by the jobs API until they
# are actually queued, so status == "queued" plus age is a clean
# "no runner has picked this up" measure. With the app token the
# scan covers the org's recently active repos; with only the
# repo-scoped GITHUB_TOKEN it covers the calling repo. A token that
# cannot read a repo's runs degrades to a warning, not a failure.
stuck=0
if [ "$MAX_QUEUE_MINUTES" -gt 0 ]; then
if [ -n "$APP_TOKEN" ]; then
GH_TOKEN="$APP_TOKEN"; scope="org"
else
GH_TOKEN="$REPO_TOKEN"; scope="repo"
if [ "$HAS_APP" = "true" ]; then
echo "::warning::App token could not be minted; falling back to a repo-scoped queue check."
fi
fi
api() {
curl -sf --max-time 10 \
-H "Accept: application/vnd.github+json" \
-H "Authorization: Bearer $GH_TOKEN" \
"$GITHUB_API_URL/$1"
}
if [ "$scope" = "org" ]; then
# Non-archived repos pushed in the last 24h, most recent first.
repos=$(api "orgs/$GITHUB_REPOSITORY_OWNER/repos?type=all&sort=pushed&direction=desc&per_page=100" \
| jq -r --argjson limit "$SCAN_REPO_LIMIT" \
'[.[] | select((.archived | not) and (.pushed_at | fromdateiso8601) > (now - 86400)) | .full_name]
| .[:$limit] | .[]') || repos=""
if [ -z "$repos" ]; then
echo "::warning::Could not list org repos with the app token; falling back to a repo-scoped queue check."
GH_TOKEN="$REPO_TOKEN"; scope="repo"
fi
fi
if [ "$scope" = "repo" ]; then
repos="$GITHUB_REPOSITORY"
fi
# One repo per line of output: "ok <count>" or "fail". Repos are
# scanned in parallel; the whole scan is bounded by the slowest
# repo rather than the sum.
scan_repo() {
local repo=$1 runs="" ids run n total=0
for status in queued in_progress; do
ids=$(api "repos/$repo/actions/runs?status=$status&per_page=20" | jq -r '.workflow_runs[].id') || { echo fail; return; }
runs="$runs $ids"
done
for run in $(echo "$runs" | tr ' ' '\n' | sort -u | sed '/^$/d'); do
n=$(api "repos/$repo/actions/runs/$run/jobs?per_page=100" | jq \
--arg primary "$PRIMARY" --argjson max "$MAX_QUEUE_MINUTES" \
'[.jobs[] | select(.status == "queued"
and (.labels | index($primary))
and (.created_at | fromdateiso8601) < (now - $max * 60))]
| length') || { echo fail; return; }
total=$((total + n))
done
echo "ok $total"
}
# Parallelism is bounded by scan-repo-limit. Each result is one
# short line, well under PIPE_BUF, so writes do not interleave.
results=$(
echo "$repos" | {
while read -r repo; do
[ -n "$repo" ] && scan_repo "$repo" &
done
wait
}
)
scanned=$(echo "$results" | grep -c '^ok' || true)
unreadable=$(echo "$results" | grep -c '^fail' || true)
stuck=$(echo "$results" | awk '/^ok/ {s += $2} END {print s + 0}')
echo "Queue scan ($scope): $scanned repo(s) scanned, $unreadable unreadable, $stuck hosted job(s) queued longer than ${MAX_QUEUE_MINUTES}m"
if [ "$scanned" -eq 0 ]; then
echo "::warning::No repo's job queue could be read (does the caller grant actions: read to this job, or pass secrets: inherit?). Skipping the queue check."
stuck=0
fi
fi
if [ "$stuck" -gt 0 ]; then
pick "$FALLBACK" "$stuck hosted job(s) queued over ${MAX_QUEUE_MINUTES}m"
else
pick "$PRIMARY" "status $actions_status"
fi