shoc-pr-review-runner/scripts/lib.sh
Adam Moussa c3cd8f7765
feat: SHOC PR review runner, phase 1
Manually-dispatched GitHub Actions workflow that reviews SHOC pull requests in
a clean environment: exact-head checkout of shoc-frontend-new and shoc-backend,
clean build/test gates, a truthful evidence report, a single-shot Fireworks
review, deterministic output validation, and published artifacts. The runner
never writes to the product repositories or their pull requests.

The review checklists move here from the reviewers' local Cursor commands so
the instructions live outside both product repos.

Phase 1 does not provision a database, start either application, or run live
browser flows; the evidence report records those as NOT_RUN so a review cannot
claim them.

Security architecture: building a PR executes its author's code, so the
workflow is split. The gates job runs that code holding no Fireworks key and
revokes its App token first; the review job holds the key, executes no product
code, and re-checks out this repo fresh. Product checkouts live outside the
workspace, the App token is downscoped at mint time, gate results fail closed
on any duplicate key, changed files are read from git objects rather than the
filesystem, and the validator re-checks every claim against the gate table.
2026-07-29 12:05:38 -04:00

105 lines
4.2 KiB
Bash
Executable file

#!/usr/bin/env bash
# Shared helpers for the SHOC PR review runner. Sourced by every script.
set -euo pipefail
: "${WORKSPACE_DIR:?WORKSPACE_DIR must be set}"
ARTIFACTS_DIR="${ARTIFACTS_DIR:-$WORKSPACE_DIR/artifacts}"
# The gate table records what actually executed, so it must NOT live inside the
# workspace: the build gates execute PR-authored code (npm/dotnet lifecycle
# scripts), and anything under the workspace is trivially writable by that code.
# STATE_DIR defaults outside the workspace; duplicate keys are rejected at read
# time so a tampered table fails the run instead of forging a PASS.
STATE_DIR="${STATE_DIR:-${RUNNER_TEMP:-$WORKSPACE_DIR}/runner-state}"
GATE_STATUS_FILE="${GATE_STATUS_FILE:-$STATE_DIR/gate-status.tsv}"
LOG_DIR="${LOG_DIR:-$ARTIFACTS_DIR/logs}"
mkdir -p "$ARTIFACTS_DIR" "$LOG_DIR" "$STATE_DIR"
chmod 700 "$STATE_DIR" 2>/dev/null || true
log() { printf '%s %s\n' "$(date -u +%H:%M:%S)" "$*" >&2; }
die() {
log "ERROR: $*"
exit 1
}
# record_gate <key> <PASS|FAIL|NOT_APPLICABLE|NOT_RUN|BLOCKED> [detail]
# Appends to the machine-readable gate table consumed by evidence generation and
# output validation. A gate is only ever PASS because the command that proves it
# actually ran and exited 0.
record_gate() {
local key="$1" status="$2" detail="${3:-}"
case "$status" in
PASS|FAIL|NOT_APPLICABLE|NOT_RUN|BLOCKED) ;;
*) die "invalid gate status '$status' for $key" ;;
esac
# Strip control characters from the detail so nothing can forge table rows.
detail="$(printf '%s' "$detail" | tr -d '\000-\037')"
printf '%s\t%s\t%s\n' "$key" "$status" "$detail" >>"$GATE_STATUS_FILE"
log "gate $key = $status${detail:+ ($detail)}"
}
# run_gate <key> <logfile-basename> <cmd...>
# Runs the command, captures combined output to the log, records PASS/FAIL.
# Returns the command's exit code so callers can decide whether to block
# dependent gates.
#
# The command may be PR-authored code, so the Actions runner-command channels
# are removed from its environment: without them it cannot append to
# $GITHUB_ENV / $GITHUB_PATH to poison later steps, or write step outputs.
run_gate() {
local key="$1" logname="$2"
shift 2
local logfile="$LOG_DIR/$logname"
log "running gate $key: $*"
local rc=0
# RUNNER_TEMP is unset too: the Actions file commands live at
# $RUNNER_TEMP/_runner_file_commands/*, so leaving it set would let the build
# re-acquire the $GITHUB_ENV / $GITHUB_PATH channel by globbing that directory.
env -u GITHUB_ENV -u GITHUB_PATH -u GITHUB_OUTPUT -u GITHUB_STATE \
-u GITHUB_STEP_SUMMARY -u ACTIONS_RUNTIME_TOKEN -u ACTIONS_ID_TOKEN_REQUEST_TOKEN \
-u ACTIONS_ID_TOKEN_REQUEST_URL -u RUNNER_TEMP -u STATE_DIR -u GATE_STATUS_FILE \
"$@" >>"$logfile" 2>&1 || rc=$?
if [ "$rc" -eq 0 ]; then
record_gate "$key" PASS "log: logs/$logname"
else
record_gate "$key" FAIL "exit $rc, log: logs/$logname"
fi
return "$rc"
}
# assert_gate_table_intact
# Fails closed if any gate key appears more than once. The runner records each
# key exactly once, so a duplicate means someone else wrote to the table. This
# catches both override-by-append and pre-seeding (writing a forged PASS for a
# key before the runner records the genuine result): either way the genuine row
# lands alongside the forged one and the duplicate is detected.
# MUST be called before any gate_status() read that informs a decision.
assert_gate_table_intact() {
[ -f "$GATE_STATUS_FILE" ] || return 0
local dupes
dupes="$(cut -f1 "$GATE_STATUS_FILE" | sort | uniq -d)"
if [ -n "$dupes" ]; then
log "duplicate gate keys detected (table tampering or a script bug):"
printf '%s\n' "$dupes" >&2
die "gate table integrity check failed"
fi
}
# gate_status <key> -> prints the recorded status, or NOT_RUN if absent.
# First-wins: the runner records each key once, so an appended row can never
# override a genuine result even if the integrity check is bypassed.
gate_status() {
local key="$1"
awk -F'\t' -v k="$key" '$1==k && !seen {s=$2; seen=1} END{print (seen?s:"NOT_RUN")}' \
"$GATE_STATUS_FILE" 2>/dev/null || echo "NOT_RUN"
}
# require_env <name>...
require_env() {
local n
for n in "$@"; do
[ -n "${!n:-}" ] || die "required environment variable $n is not set"
done
}