Adds SYNC_WEB=1 to /sh-deploy-r720's script: build agent-team/web/dist on the Mac (box Node too old for Vite), rsync it with --delete (clears stale hashed bundles), force a status-service restart, and verify the served index.html references the freshly-built bundle. Also stops the verify step false-failing on the known benign draft-pr-monitor 'gh'-missing traceback (strips that block before judging the journal; is-active/NRestarts/threads remain the authoritative crash-loop signals).
183 lines
10 KiB
Bash
Executable file
183 lines
10 KiB
Bash
Executable file
#!/usr/bin/env bash
|
|
# deploy-r720.sh — SAFE, idempotent UPDATE of the live R720 agent-team coordinator.
|
|
#
|
|
# This is the GENERAL deploy path for ongoing agent-team code changes (the
|
|
# worked, one-off example is deploy-r720-ws-rollout.sh). It is deploy-before-merge:
|
|
# rsync the WHOLE agent_team package from your local checkout, restart the
|
|
# coordinator, and VERIFY. Run it from the MAC.
|
|
#
|
|
# It exists because of two self-inflicted live crash-loops:
|
|
# 1) Per-file rsync drops the subdir and lands e.g. nodes/planner.py at the
|
|
# package root -> ImportError/TypeError crash-loop. This script rsyncs the
|
|
# WHOLE agent_team/ tree (structure preserved), never individual files.
|
|
# 2) A bad deploy reads as `activating` (systemd auto-restart), NOT a clean
|
|
# error, and the daemon logs WARNING+ only (empty journal != healthy). So
|
|
# this script VERIFIES is-active==active + NRestarts-didn't-climb + thread
|
|
# count ~6 after the restart, and tells you to roll back via the snapshot
|
|
# on failure.
|
|
#
|
|
# The Hyper-V snapshot is Adam's step on the WINDOWS HYPERVISOR (10.10.60.40) —
|
|
# Claude's `ssh secrev` reaches only the guest VM. This script PROMPTS to
|
|
# confirm the snapshot and refuses to proceed without it.
|
|
#
|
|
# It is idempotent and FAILS LOUDLY (set -euo pipefail).
|
|
#
|
|
# Usage:
|
|
# REPO=~/Documents/repositories/orchestrator bash agent-team/scripts/deploy-r720.sh
|
|
# Options (env):
|
|
# SYNC_DEPS=1 also rsync requirements.txt + pip install into the box venv
|
|
# (set this whenever requirements.txt changed — esp. the
|
|
# langchain-* model stack, or models.py fails to import and
|
|
# the non-Claude review/scan/build silently never run).
|
|
# SYNC_HANDBOOK=1 also rsync the engineering handbook (WS5 context_provider).
|
|
# SYNC_WEB=1 also (re)deploy the dashboard SPA: build web/dist ON THE MAC
|
|
# (the box Node is too old for Vite 5+), rsync it with --delete
|
|
# (clears stale content-hashed bundles), and force a status-
|
|
# service restart + a served-bundle verify. Set this whenever
|
|
# agent-team/web changed.
|
|
# RESTART_STATUS=1 also restart agent-team-status.service (the LAN dashboard).
|
|
# Implied by SYNC_WEB=1.
|
|
# ASSUME_SNAPSHOT=1 skip the interactive snapshot prompt (you confirmed it
|
|
# out of band). Do NOT set this casually.
|
|
set -euo pipefail
|
|
|
|
# ── Config (override via env) ────────────────────────────────────────────────
|
|
BOX="${BOX:-secrev}" # ssh alias -> guest VM (key r720_seahaven, NOPASSWD sudo)
|
|
REPO="${REPO:-$HOME/Documents/repositories/orchestrator}"
|
|
HANDBOOK_LOCAL="${HANDBOOK_LOCAL:-$HOME/Documents/repositories/engineering-handbook}"
|
|
HANDBOOK_REMOTE="${HANDBOOK_REMOTE:-.sea-haven/engineering-handbook}" # matches SEA_HAVEN_HANDBOOK_DIR
|
|
DATE="${DATE:-$(date +%Y%m%d)}"
|
|
SVC="agent-team-coordinator.service"
|
|
SSH="ssh ${BOX}"
|
|
|
|
say() { printf '\n\033[1;36m== %s\033[0m\n' "$*"; }
|
|
err() { printf '\033[1;31m!! %s\033[0m\n' "$*" >&2; }
|
|
confirm() { read -r -p "$1 [y/N] " a; [ "$a" = "y" ] || [ "$a" = "Y" ]; }
|
|
|
|
# ── 0. Pre-flight: reachability + the snapshot HARD GATE ─────────────────────
|
|
say "0. Pre-flight"
|
|
[ -d "${REPO}/agent-team/agent_team" ] || { err "no agent_team package under REPO=${REPO}"; exit 1; }
|
|
$SSH true || { err "cannot reach box (${BOX}); check the secrev alias / r720_seahaven key"; exit 1; }
|
|
|
|
echo "SNAPSHOT is Adam's step on the Hyper-V HOST (Claude's ssh reaches only the guest VM):"
|
|
echo " ssh Administrator@10.10.60.40 # PowerShell"
|
|
echo " Checkpoint-VM -Name sh-secrev -SnapshotName pre-deploy-${DATE}"
|
|
if [ "${ASSUME_SNAPSHOT:-0}" != "1" ]; then
|
|
confirm "Hyper-V snapshot taken and confirmed?" || { err "aborted: snapshot not confirmed"; exit 1; }
|
|
fi
|
|
|
|
# Record current health to compare after the restart.
|
|
read -r NRESTARTS_BEFORE < <($SSH "systemctl show ${SVC} -p NRestarts --value")
|
|
echo "current NRestarts=${NRESTARTS_BEFORE}, ActiveState=$($SSH "systemctl show ${SVC} -p ActiveState --value")"
|
|
|
|
# ── 1. Back up the ledger BEFORE any change ──────────────────────────────────
|
|
say "1. Back up the ledger"
|
|
$SSH "cp ~/orchestrator/agent-team/state/agent_team.sqlite ~/agent_team.sqlite.bak-${DATE}"
|
|
echo "ledger -> ~/agent_team.sqlite.bak-${DATE}"
|
|
|
|
# ── 2. Sync code the SAFE way: WHOLE package, never individual files ─────────
|
|
say "2. rsync the WHOLE agent_team package (structure preserved) + run-team.py"
|
|
rsync -az --exclude '__pycache__' --exclude '*.pyc' --exclude 'state' \
|
|
"${REPO}/agent-team/agent_team/" "${BOX}:orchestrator/agent-team/agent_team/"
|
|
rsync -az "${REPO}/agent-team/run-team.py" "${BOX}:orchestrator/agent-team/run-team.py"
|
|
|
|
if [ "${SYNC_DEPS:-0}" = "1" ]; then
|
|
say "2a. requirements changed -> rsync + reinstall into the box venv"
|
|
rsync -az "${REPO}/requirements.txt" "${BOX}:orchestrator/requirements.txt"
|
|
$SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && pip install --upgrade -r ~/orchestrator/requirements.txt'
|
|
fi
|
|
|
|
if [ "${SYNC_HANDBOOK:-0}" = "1" ]; then
|
|
say "2b. sync engineering handbook (WS5 context_provider source)"
|
|
if [ -d "${HANDBOOK_LOCAL}" ]; then
|
|
$SSH "mkdir -p ${HANDBOOK_REMOTE}"
|
|
rsync -az --delete --exclude .git "${HANDBOOK_LOCAL}/" "${BOX}:${HANDBOOK_REMOTE}/"
|
|
else
|
|
err "HANDBOOK_LOCAL=${HANDBOOK_LOCAL} not found; skipping (context_provider fail-safes to '')."
|
|
fi
|
|
fi
|
|
|
|
if [ "${SYNC_WEB:-0}" = "1" ]; then
|
|
say "2d. build the dashboard SPA on the Mac + rsync web/dist (box Node too old for Vite)"
|
|
WEB="${REPO}/agent-team/web"
|
|
[ -d "${WEB}" ] || { err "no web/ dir under ${REPO}/agent-team"; exit 1; }
|
|
# Build on the Mac (npm ci only if deps are missing), then ship the whole dist
|
|
# with --delete so old hashed bundles don't linger on the box.
|
|
( cd "${WEB}" && { [ -d node_modules ] || npm ci; } && npm run build )
|
|
rsync -az --delete "${WEB}/dist/" "${BOX}:orchestrator/agent-team/web/dist/"
|
|
RESTART_STATUS=1 # the dashboard must restart to serve the new bundle
|
|
fi
|
|
|
|
# ── 2c. Pre-restart import sanity (catch the crash before it loops) ──────────
|
|
say "2c. Pre-restart import sanity"
|
|
$SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && \
|
|
python3 -c "import agent_team.coordinator, agent_team.graph; from agent_team.nodes import planner; print(\"imports OK\")"' \
|
|
|| { err "import check FAILED — the sync is broken (likely a misplaced module). Fix the sync; do NOT restart."; exit 1; }
|
|
|
|
# ── 3. Restart ───────────────────────────────────────────────────────────────
|
|
say "3. Restart ${SVC}"
|
|
confirm "Restart the LIVE coordinator now?" || { err "skipped restart"; exit 0; }
|
|
$SSH "sudo systemctl restart ${SVC}"
|
|
if [ "${RESTART_STATUS:-0}" = "1" ]; then
|
|
$SSH 'sudo systemctl restart agent-team-status.service'
|
|
fi
|
|
|
|
# ── 4. VERIFY (mandatory) ────────────────────────────────────────────────────
|
|
say "4. Verify (is-active / NRestarts / threads / journal)"
|
|
sleep 3
|
|
STATE=$($SSH "systemctl is-active ${SVC}" || true)
|
|
read -r NRESTARTS_AFTER < <($SSH "systemctl show ${SVC} -p NRestarts --value")
|
|
MAINPID=$($SSH "systemctl show ${SVC} -p MainPID --value")
|
|
THREADS=$($SSH "ls /proc/${MAINPID}/task 2>/dev/null | wc -l" || echo 0)
|
|
echo "is-active=${STATE} NRestarts ${NRESTARTS_BEFORE} -> ${NRESTARTS_AFTER} MainPID=${MAINPID} threads=${THREADS}"
|
|
|
|
OK=1
|
|
[ "${STATE}" = "active" ] || { err "is-active=${STATE} (expected active; 'activating' = crash-loop)"; OK=0; }
|
|
[ "${NRESTARTS_AFTER}" -le "${NRESTARTS_BEFORE}" ] || { err "NRestarts climbed (${NRESTARTS_BEFORE}->${NRESTARTS_AFTER}) = crash-loop"; OK=0; }
|
|
[ "${THREADS}" -ge 5 ] || { err "thread count ${THREADS} (~6 expected; Socket Mode listener may be down)"; OK=0; }
|
|
# Strip the KNOWN-BENIGN draft-pr-monitor 'gh'-missing traceback (the box lacks the
|
|
# gh CLI; it fail-safes and fires every ~30s) before judging the journal, else every
|
|
# deploy false-fails. See feedback_r720_agent_team_deploy.
|
|
JOURNAL="$($SSH "journalctl -u ${SVC} --since '30 seconds ago' --no-pager")"
|
|
CLEAN="$(printf '%s\n' "${JOURNAL}" | sed "/Traceback (most recent call last):/,/No such file or directory: 'gh'/d")"
|
|
if printf '%s\n' "${CLEAN}" | grep -Eiq 'traceback|error'; then
|
|
err "journal shows a (non-'gh') Traceback/error in the last 30s"; OK=0
|
|
fi
|
|
|
|
if [ "${OK}" != "1" ]; then
|
|
err "DEPLOY VERIFY FAILED — rolling back is required."
|
|
echo "Evidence:"; $SSH "journalctl -u ${SVC} -n 80 --no-pager" || true
|
|
cat <<ROLLBACK_MSG
|
|
|
|
ROLLBACK (Adam, on the Hyper-V host 10.10.60.40):
|
|
Restore-VMSnapshot -Name pre-deploy-${DATE} -VMName sh-secrev -Confirm:\$false
|
|
The ledger backup ~/agent_team.sqlite.bak-${DATE} is the secondary safety net.
|
|
Re-verify health after rollback before doing anything else.
|
|
ROLLBACK_MSG
|
|
exit 1
|
|
fi
|
|
|
|
say "VERIFY PASSED — coordinator active, no restart climb, listener up."
|
|
|
|
# ── 4b. Frontend verify (only when the dashboard service was (re)started) ─────
|
|
if [ "${RESTART_STATUS:-0}" = "1" ]; then
|
|
say "4b. Verify the dashboard (agent-team-status.service)"
|
|
sleep 2
|
|
SSTATE=$($SSH "systemctl is-active agent-team-status.service" || true)
|
|
HTTP=$($SSH "curl -s -o /dev/null -w '%{http_code}' http://127.0.0.1:8770/" || echo 000)
|
|
echo "status-service is-active=${SSTATE} dashboard http=${HTTP}"
|
|
[ "${SSTATE}" = "active" ] || err "agent-team-status.service is ${SSTATE} (expected active)"
|
|
[ "${HTTP}" = "200" ] || err "dashboard returned http ${HTTP} (expected 200)"
|
|
if [ "${SYNC_WEB:-0}" = "1" ]; then
|
|
# Confirm the box actually serves the bundle we just built (no stale dist).
|
|
BUNDLE=$(grep -oE 'assets/index-[A-Za-z0-9_-]+\.js' "${REPO}/agent-team/web/dist/index.html" | head -1)
|
|
if [ -n "${BUNDLE}" ] && $SSH "curl -s http://127.0.0.1:8770/" | grep -q "${BUNDLE}"; then
|
|
echo "served index.html references the freshly-built ${BUNDLE} ✓"
|
|
else
|
|
err "served index.html does NOT reference the new bundle (${BUNDLE:-?}) — stale dist?"
|
|
fi
|
|
fi
|
|
fi
|
|
|
|
echo "Reminder: merges to main remain held (deploy-before-merge)."
|
|
say "Done."
|