shoc-pr-review-runner/scripts/validate-review-output.sh
Adam Moussa c3cd8f7765
feat: SHOC PR review runner, phase 1
Manually-dispatched GitHub Actions workflow that reviews SHOC pull requests in
a clean environment: exact-head checkout of shoc-frontend-new and shoc-backend,
clean build/test gates, a truthful evidence report, a single-shot Fireworks
review, deterministic output validation, and published artifacts. The runner
never writes to the product repositories or their pull requests.

The review checklists move here from the reviewers' local Cursor commands so
the instructions live outside both product repos.

Phase 1 does not provision a database, start either application, or run live
browser flows; the evidence report records those as NOT_RUN so a review cannot
claim them.

Security architecture: building a PR executes its author's code, so the
workflow is split. The gates job runs that code holding no Fireworks key and
revokes its App token first; the review job holds the key, executes no product
code, and re-checks out this repo fresh. Product checkouts live outside the
workspace, the App token is downscoped at mint time, gate results fail closed
on any duplicate key, changed files are read from git objects rather than the
filesystem, and the validator re-checks every claim against the gate table.
2026-07-29 12:05:38 -04:00

203 lines
8.6 KiB
Bash
Executable file

#!/usr/bin/env bash
# Deterministic validation of the generated review (spec §20).
#
# This is the backstop that does not trust the model: the PR content in the
# prompt is attacker-influenceable, so every property that matters is re-checked
# here against the recorded gate table rather than against what the review says.
#
# Usage: validate-review-output.sh <review-file>
# Reads: REVIEW_TYPE, ARTIFACTS_DIR (<side>-pr.json), gate table via lib.sh.
# Prints each violation on its own line; exits non-zero if any is found.
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=lib.sh
source "$SCRIPT_DIR/lib.sh"
review="${1:?review file required}"
[ -f "$review" ] || { echo "VALIDATION: review file '$review' does not exist"; exit 1; }
require_env REVIEW_TYPE
errors=0
err() { echo "VALIDATION: $*"; errors=$((errors + 1)); }
side_in_scope() {
case "$REVIEW_TYPE" in
paired) return 0 ;;
"$1") return 0 ;;
*) return 1 ;;
esac
}
# A tampered or duplicated gate table invalidates every decision below.
assert_gate_table_intact
# --- headings: each exactly once, in order -----------------------------------
count_heading() { grep -c "^### $1\\. " "$review" || true; }
for n in 1 2 3; do
c="$(count_heading "$n")"
[ "$c" -eq 1 ] || err "heading '### $n.' must appear exactly once (found $c)"
done
h1="$(grep -n '^### 1\. Overall Verdict' "$review" | head -1 | cut -d: -f1 || true)"
h2="$(grep -n '^### 2\. Overall Review Comment' "$review" | head -1 | cut -d: -f1 || true)"
h3="$(grep -n '^### 3\. Inline Comments' "$review" | head -1 | cut -d: -f1 || true)"
[ -n "$h1" ] || err "missing heading '### 1. Overall Verdict'"
[ -n "$h2" ] || err "missing heading '### 2. Overall Review Comment'"
[ -n "$h3" ] || err "missing heading '### 3. Inline Comments'"
if [ -n "$h1" ] && [ -n "$h2" ] && [ -n "$h3" ]; then
{ [ "$h1" -lt "$h2" ] && [ "$h2" -lt "$h3" ]; } || err "headings are out of order"
fi
# --- verdict: exactly one anchored verdict line, no stray verdict tokens ------
verdict=""
if [ -n "$h1" ] && [ -n "$h2" ]; then
verdict_block="$(sed -n "$((h1 + 1)),$((h2 - 1))p" "$review")"
# Anchored: optional backticks, the token, then " - " and a reason.
# shellcheck disable=SC2016 # backticks are literal markdown
verdict_lines="$(printf '%s\n' "$verdict_block" \
| grep -cE '^[[:space:]]*`?(APPROVE|REQUEST_CHANGES|COMMENT)`?[[:space:]]+-[[:space:]]+' || true)"
if [ "$verdict_lines" -ne 1 ]; then
err "section 1 must contain exactly one verdict line of the form '\`VERDICT\` - reason' (found $verdict_lines)"
fi
# Any additional verdict token anywhere in section 1 is a laundering attempt.
token_count="$(printf '%s\n' "$verdict_block" | grep -oE 'APPROVE|REQUEST_CHANGES|COMMENT' | wc -l | tr -d ' ')"
if [ "$token_count" -gt 1 ]; then
err "section 1 contains $token_count verdict tokens; exactly one is allowed"
fi
# shellcheck disable=SC2016 # backticks are literal markdown
verdict="$(printf '%s\n' "$verdict_block" \
| grep -oE '^[[:space:]]*`?(APPROVE|REQUEST_CHANGES|COMMENT)`?[[:space:]]+-' \
| grep -oE 'APPROVE|REQUEST_CHANGES|COMMENT' | head -1 || true)"
[ -n "$verdict" ] || err "no valid anchored verdict in section 1"
fi
# --- SHA rules ---------------------------------------------------------------
if grep -qE '[0-9a-f]{40}' "$review"; then
err "full 40-character commit SHA present — outward-facing copy must use the 7-character SHA only"
fi
for side in frontend backend; do
if side_in_scope "$side" && [ -f "$ARTIFACTS_DIR/$side-pr.json" ]; then
short="$(jq -r '.short_sha' "$ARTIFACTS_DIR/$side-pr.json")"
grep -q "$short" "$review" || err "review does not mention the reviewed $side head SHA $short"
fi
done
# --- inline comments: per-blocker Fix/Test accounting ------------------------
blockers=0
if [ -n "$h3" ]; then
inline_block="$(sed -n "$((h3 + 1)),\$p" "$review")"
# shellcheck disable=SC2016 # backticks are literal markdown
blocker_re='^\*\*`[^`]+:[0-9]+`\*\*'
blockers="$(printf '%s\n' "$inline_block" | grep -cE "$blocker_re" || true)"
if [ "$blockers" -gt 0 ]; then
# Split into per-blocker chunks with awk and require exactly one Fix and one
# Test inside each, so a doubled pair cannot cover a bare blocker.
bad="$(printf '%s\n' "$inline_block" | awk -v re="$blocker_re" '
function flush() {
if (started) {
if (fix != 1 || test != 1)
printf "%s (Fix:%d Test:%d)\n", header, fix, test
}
}
$0 ~ re { flush(); started=1; header=$0; fix=0; test=0; next }
started && /^\*\*Fix:\*\*/ { fix++ }
started && /^\*\*Test:\*\*/ { test++ }
!started && (/^\*\*Fix:\*\*/ || /^\*\*Test:\*\*/) { print "Fix/Test line before the first blocker" }
END { flush() }')"
if [ -n "$bad" ]; then
while IFS= read -r line; do
[ -n "$line" ] && err "each inline blocker needs exactly one Fix and one Test line: $line"
done <<<"$bad"
fi
else
printf '%s\n' "$inline_block" | grep -q '^None\.$' \
|| err "inline comments must contain at least one blocker or exactly 'None.'"
fi
fi
case "$verdict" in
REQUEST_CHANGES)
[ "$blockers" -gt 0 ] || err "REQUEST_CHANGES verdict requires at least one inline blocker with file path and line"
;;
APPROVE)
[ "$blockers" -eq 0 ] || err "APPROVE verdict must not carry inline blockers"
;;
esac
# --- required gates ----------------------------------------------------------
required=()
if side_in_scope frontend; then
required+=(frontend.install frontend.lint frontend.build frontend.unit_tests)
fi
if side_in_scope backend; then
required+=(backend.restore backend.build backend.test)
fi
if [ "$verdict" = "APPROVE" ]; then
for gate in "${required[@]}"; do
s="$(gate_status "$gate")"
[ "$s" = "PASS" ] || err "APPROVE is forbidden while required gate '$gate' is $s"
done
if side_in_scope frontend; then
case "$(gate_status frontend.e2e_mocked)" in
FAIL|BLOCKED) err "APPROVE is forbidden while the mocked Playwright gate is $(gate_status frontend.e2e_mocked)" ;;
esac
fi
fi
# --- per-gate claim check (applies to EVERY verdict) -------------------------
# A review must not describe a gate as clean when the table says otherwise.
# Keyed on the gate's noun appearing near a positive-result word.
check_claim() { # check_claim <gate-key> <noun-regex>
local gate="$1" noun="$2" status
status="$(gate_status "$gate")"
if [ "$status" = "PASS" ]; then return 0; fi
if grep -qiE "${noun}[^.]{0,60}(clean|green|pass(es|ed|ing)?|succeed(s|ed)?|successful|no (errors|failures)|all good)" "$review" \
|| grep -qiE "(clean|green|passing|successful|no (errors|failures))[^.]{0,60}${noun}" "$review"; then
err "review describes '$noun' as clean but gate '$gate' is $status"
fi
}
if side_in_scope frontend; then
check_claim frontend.install '(npm ci|install|dependencies)'
check_claim frontend.lint 'lint'
check_claim frontend.build '(build|typescript|tsc|compile)'
check_claim frontend.unit_tests '(unit tests?|vitest|component tests?)'
check_claim frontend.e2e_mocked '(playwright|e2e|end.to.end|browser tests?)'
fi
if side_in_scope backend; then
check_claim backend.restore '(restore|nuget)'
check_claim backend.build '(build|compile|release build)'
check_claim backend.test '(tests?|xunit)'
fi
# --- unsupported claims ------------------------------------------------------
# Phase 1 never starts the applications or runs live browser flows, so these
# claims can never be supported by evidence.
if grep -qiE 'verified in the browser|live (browser|integration) (coverage|validation|tests?|suite) (passed|succeeded|is clean)' "$review"; then
err "review claims live browser validation, which was NOT_RUN"
fi
if grep -qiE 'api contract (is|was) (correct|verified)' "$review"; then
err "review claims runtime API contract verification, which was NOT_RUN (mock/static inspection only)"
fi
if grep -qiE '(application|api|backend|frontend) (started|starts|is running|was running)' "$review"; then
err "review claims the application was started, which was NOT_RUN in Phase 1"
fi
if grep -qiE 'all tests pass(ed)?' "$review"; then
for gate in "${required[@]}"; do
case "$gate" in
*test*) [ "$(gate_status "$gate")" = "PASS" ] || err "review claims all tests passed but gate '$gate' is $(gate_status "$gate")" ;;
esac
done
fi
# Leftover extraction delimiters mean the review body was spliced.
if grep -qE '</?REVIEW[^>]*>' "$review"; then
err "review body contains extraction delimiters — output was spliced from multiple blocks"
fi
if [ "$errors" -gt 0 ]; then
echo "VALIDATION FAILED: $errors error(s)"
exit 1
fi
echo "VALIDATION PASSED"