Merge pull request #50 from Sea-Haven-Industries/feat/r720-deploy-script
feat(agent-team): generalized SAFE deploy-r720.sh (whole-package rsync, snapshot-gate, verify-after)
This commit is contained in:
commit
e8369e3c1d
1 changed files with 140 additions and 0 deletions
140
agent-team/scripts/deploy-r720.sh
Executable file
140
agent-team/scripts/deploy-r720.sh
Executable file
|
|
@ -0,0 +1,140 @@
|
|||
#!/usr/bin/env bash
|
||||
# deploy-r720.sh — SAFE, idempotent UPDATE of the live R720 agent-team coordinator.
|
||||
#
|
||||
# This is the GENERAL deploy path for ongoing agent-team code changes (the
|
||||
# worked, one-off example is deploy-r720-ws-rollout.sh). It is deploy-before-merge:
|
||||
# rsync the WHOLE agent_team package from your local checkout, restart the
|
||||
# coordinator, and VERIFY. Run it from the MAC.
|
||||
#
|
||||
# It exists because of two self-inflicted live crash-loops:
|
||||
# 1) Per-file rsync drops the subdir and lands e.g. nodes/planner.py at the
|
||||
# package root -> ImportError/TypeError crash-loop. This script rsyncs the
|
||||
# WHOLE agent_team/ tree (structure preserved), never individual files.
|
||||
# 2) A bad deploy reads as `activating` (systemd auto-restart), NOT a clean
|
||||
# error, and the daemon logs WARNING+ only (empty journal != healthy). So
|
||||
# this script VERIFIES is-active==active + NRestarts-didn't-climb + thread
|
||||
# count ~6 after the restart, and tells you to roll back via the snapshot
|
||||
# on failure.
|
||||
#
|
||||
# The Hyper-V snapshot is Adam's step on the WINDOWS HYPERVISOR (10.10.60.40) —
|
||||
# Claude's `ssh secrev` reaches only the guest VM. This script PROMPTS to
|
||||
# confirm the snapshot and refuses to proceed without it.
|
||||
#
|
||||
# It is idempotent and FAILS LOUDLY (set -euo pipefail).
|
||||
#
|
||||
# Usage:
|
||||
# REPO=~/Documents/repositories/orchestrator bash agent-team/scripts/deploy-r720.sh
|
||||
# Options (env):
|
||||
# SYNC_DEPS=1 also rsync requirements.txt + pip install into the box venv
|
||||
# (set this whenever requirements.txt changed — esp. the
|
||||
# langchain-* model stack, or models.py fails to import and
|
||||
# the non-Claude review/scan/build silently never run).
|
||||
# SYNC_HANDBOOK=1 also rsync the engineering handbook (WS5 context_provider).
|
||||
# RESTART_STATUS=1 also restart agent-team-status.service (the LAN dashboard).
|
||||
# ASSUME_SNAPSHOT=1 skip the interactive snapshot prompt (you confirmed it
|
||||
# out of band). Do NOT set this casually.
|
||||
set -euo pipefail
|
||||
|
||||
# ── Config (override via env) ────────────────────────────────────────────────
|
||||
BOX="${BOX:-secrev}" # ssh alias -> guest VM (key r720_seahaven, NOPASSWD sudo)
|
||||
REPO="${REPO:-$HOME/Documents/repositories/orchestrator}"
|
||||
HANDBOOK_LOCAL="${HANDBOOK_LOCAL:-$HOME/Documents/repositories/engineering-handbook}"
|
||||
HANDBOOK_REMOTE="${HANDBOOK_REMOTE:-.sea-haven/engineering-handbook}" # matches SEA_HAVEN_HANDBOOK_DIR
|
||||
DATE="${DATE:-$(date +%Y%m%d)}"
|
||||
SVC="agent-team-coordinator.service"
|
||||
SSH="ssh ${BOX}"
|
||||
|
||||
say() { printf '\n\033[1;36m== %s\033[0m\n' "$*"; }
|
||||
err() { printf '\033[1;31m!! %s\033[0m\n' "$*" >&2; }
|
||||
confirm() { read -r -p "$1 [y/N] " a; [ "$a" = "y" ] || [ "$a" = "Y" ]; }
|
||||
|
||||
# ── 0. Pre-flight: reachability + the snapshot HARD GATE ─────────────────────
|
||||
say "0. Pre-flight"
|
||||
[ -d "${REPO}/agent-team/agent_team" ] || { err "no agent_team package under REPO=${REPO}"; exit 1; }
|
||||
$SSH true || { err "cannot reach box (${BOX}); check the secrev alias / r720_seahaven key"; exit 1; }
|
||||
|
||||
echo "SNAPSHOT is Adam's step on the Hyper-V HOST (Claude's ssh reaches only the guest VM):"
|
||||
echo " ssh Administrator@10.10.60.40 # PowerShell"
|
||||
echo " Checkpoint-VM -Name sh-secrev -SnapshotName pre-deploy-${DATE}"
|
||||
if [ "${ASSUME_SNAPSHOT:-0}" != "1" ]; then
|
||||
confirm "Hyper-V snapshot taken and confirmed?" || { err "aborted: snapshot not confirmed"; exit 1; }
|
||||
fi
|
||||
|
||||
# Record current health to compare after the restart.
|
||||
read -r NRESTARTS_BEFORE < <($SSH "systemctl show ${SVC} -p NRestarts --value")
|
||||
echo "current NRestarts=${NRESTARTS_BEFORE}, ActiveState=$($SSH "systemctl show ${SVC} -p ActiveState --value")"
|
||||
|
||||
# ── 1. Back up the ledger BEFORE any change ──────────────────────────────────
|
||||
say "1. Back up the ledger"
|
||||
$SSH "cp ~/orchestrator/agent-team/state/agent_team.sqlite ~/agent_team.sqlite.bak-${DATE}"
|
||||
echo "ledger -> ~/agent_team.sqlite.bak-${DATE}"
|
||||
|
||||
# ── 2. Sync code the SAFE way: WHOLE package, never individual files ─────────
|
||||
say "2. rsync the WHOLE agent_team package (structure preserved) + run-team.py"
|
||||
rsync -az --exclude '__pycache__' --exclude '*.pyc' --exclude 'state' \
|
||||
"${REPO}/agent-team/agent_team/" "${BOX}:orchestrator/agent-team/agent_team/"
|
||||
rsync -az "${REPO}/agent-team/run-team.py" "${BOX}:orchestrator/agent-team/run-team.py"
|
||||
|
||||
if [ "${SYNC_DEPS:-0}" = "1" ]; then
|
||||
say "2a. requirements changed -> rsync + reinstall into the box venv"
|
||||
rsync -az "${REPO}/requirements.txt" "${BOX}:orchestrator/requirements.txt"
|
||||
$SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && pip install --upgrade -r ~/orchestrator/requirements.txt'
|
||||
fi
|
||||
|
||||
if [ "${SYNC_HANDBOOK:-0}" = "1" ]; then
|
||||
say "2b. sync engineering handbook (WS5 context_provider source)"
|
||||
if [ -d "${HANDBOOK_LOCAL}" ]; then
|
||||
$SSH "mkdir -p ${HANDBOOK_REMOTE}"
|
||||
rsync -az --delete --exclude .git "${HANDBOOK_LOCAL}/" "${BOX}:${HANDBOOK_REMOTE}/"
|
||||
else
|
||||
err "HANDBOOK_LOCAL=${HANDBOOK_LOCAL} not found; skipping (context_provider fail-safes to '')."
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── 2c. Pre-restart import sanity (catch the crash before it loops) ──────────
|
||||
say "2c. Pre-restart import sanity"
|
||||
$SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && \
|
||||
python3 -c "import agent_team.coordinator, agent_team.graph; from agent_team.nodes import planner; print(\"imports OK\")"' \
|
||||
|| { err "import check FAILED — the sync is broken (likely a misplaced module). Fix the sync; do NOT restart."; exit 1; }
|
||||
|
||||
# ── 3. Restart ───────────────────────────────────────────────────────────────
|
||||
say "3. Restart ${SVC}"
|
||||
confirm "Restart the LIVE coordinator now?" || { err "skipped restart"; exit 0; }
|
||||
$SSH "sudo systemctl restart ${SVC}"
|
||||
if [ "${RESTART_STATUS:-0}" = "1" ]; then
|
||||
$SSH 'sudo systemctl restart agent-team-status.service'
|
||||
fi
|
||||
|
||||
# ── 4. VERIFY (mandatory) ────────────────────────────────────────────────────
|
||||
say "4. Verify (is-active / NRestarts / threads / journal)"
|
||||
sleep 3
|
||||
STATE=$($SSH "systemctl is-active ${SVC}" || true)
|
||||
read -r NRESTARTS_AFTER < <($SSH "systemctl show ${SVC} -p NRestarts --value")
|
||||
MAINPID=$($SSH "systemctl show ${SVC} -p MainPID --value")
|
||||
THREADS=$($SSH "ls /proc/${MAINPID}/task 2>/dev/null | wc -l" || echo 0)
|
||||
echo "is-active=${STATE} NRestarts ${NRESTARTS_BEFORE} -> ${NRESTARTS_AFTER} MainPID=${MAINPID} threads=${THREADS}"
|
||||
|
||||
OK=1
|
||||
[ "${STATE}" = "active" ] || { err "is-active=${STATE} (expected active; 'activating' = crash-loop)"; OK=0; }
|
||||
[ "${NRESTARTS_AFTER}" -le "${NRESTARTS_BEFORE}" ] || { err "NRestarts climbed (${NRESTARTS_BEFORE}->${NRESTARTS_AFTER}) = crash-loop"; OK=0; }
|
||||
[ "${THREADS}" -ge 5 ] || { err "thread count ${THREADS} (~6 expected; Socket Mode listener may be down)"; OK=0; }
|
||||
if $SSH "journalctl -u ${SVC} --since '30 seconds ago' --no-pager" | grep -Eiq 'traceback|error'; then
|
||||
err "journal shows Traceback/error in the last 30s"; OK=0
|
||||
fi
|
||||
|
||||
if [ "${OK}" != "1" ]; then
|
||||
err "DEPLOY VERIFY FAILED — rolling back is required."
|
||||
echo "Evidence:"; $SSH "journalctl -u ${SVC} -n 80 --no-pager" || true
|
||||
cat <<ROLLBACK_MSG
|
||||
|
||||
ROLLBACK (Adam, on the Hyper-V host 10.10.60.40):
|
||||
Restore-VMSnapshot -Name pre-deploy-${DATE} -VMName sh-secrev -Confirm:\$false
|
||||
The ledger backup ~/agent_team.sqlite.bak-${DATE} is the secondary safety net.
|
||||
Re-verify health after rollback before doing anything else.
|
||||
ROLLBACK_MSG
|
||||
exit 1
|
||||
fi
|
||||
|
||||
say "VERIFY PASSED — coordinator active, no restart climb, listener up."
|
||||
echo "Reminder: merges to main remain held (deploy-before-merge)."
|
||||
say "Done."
|
||||
Reference in a new issue