This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/scripts/deploy-r720.sh

141 lines
7.8 KiB
Bash
Raw Normal View History

#!/usr/bin/env bash
# deploy-r720.sh — SAFE, idempotent UPDATE of the live R720 agent-team coordinator.
#
# This is the GENERAL deploy path for ongoing agent-team code changes (the
# worked, one-off example is deploy-r720-ws-rollout.sh). It is deploy-before-merge:
# rsync the WHOLE agent_team package from your local checkout, restart the
# coordinator, and VERIFY. Run it from the MAC.
#
# It exists because of two self-inflicted live crash-loops:
# 1) Per-file rsync drops the subdir and lands e.g. nodes/planner.py at the
# package root -> ImportError/TypeError crash-loop. This script rsyncs the
# WHOLE agent_team/ tree (structure preserved), never individual files.
# 2) A bad deploy reads as `activating` (systemd auto-restart), NOT a clean
# error, and the daemon logs WARNING+ only (empty journal != healthy). So
# this script VERIFIES is-active==active + NRestarts-didn't-climb + thread
# count ~6 after the restart, and tells you to roll back via the snapshot
# on failure.
#
# The Hyper-V snapshot is Adam's step on the WINDOWS HYPERVISOR (10.10.60.40) —
# Claude's `ssh secrev` reaches only the guest VM. This script PROMPTS to
# confirm the snapshot and refuses to proceed without it.
#
# It is idempotent and FAILS LOUDLY (set -euo pipefail).
#
# Usage:
# REPO=~/Documents/repositories/orchestrator bash agent-team/scripts/deploy-r720.sh
# Options (env):
# SYNC_DEPS=1 also rsync requirements.txt + pip install into the box venv
# (set this whenever requirements.txt changed — esp. the
# langchain-* model stack, or models.py fails to import and
# the non-Claude review/scan/build silently never run).
# SYNC_HANDBOOK=1 also rsync the engineering handbook (WS5 context_provider).
# RESTART_STATUS=1 also restart agent-team-status.service (the LAN dashboard).
# ASSUME_SNAPSHOT=1 skip the interactive snapshot prompt (you confirmed it
# out of band). Do NOT set this casually.
set -euo pipefail
# ── Config (override via env) ────────────────────────────────────────────────
BOX="${BOX:-secrev}" # ssh alias -> guest VM (key r720_seahaven, NOPASSWD sudo)
REPO="${REPO:-$HOME/Documents/repositories/orchestrator}"
HANDBOOK_LOCAL="${HANDBOOK_LOCAL:-$HOME/Documents/repositories/engineering-handbook}"
HANDBOOK_REMOTE="${HANDBOOK_REMOTE:-.sea-haven/engineering-handbook}" # matches SEA_HAVEN_HANDBOOK_DIR
DATE="${DATE:-$(date +%Y%m%d)}"
SVC="agent-team-coordinator.service"
SSH="ssh ${BOX}"
say() { printf '\n\033[1;36m== %s\033[0m\n' "$*"; }
err() { printf '\033[1;31m!! %s\033[0m\n' "$*" >&2; }
confirm() { read -r -p "$1 [y/N] " a; [ "$a" = "y" ] || [ "$a" = "Y" ]; }
# ── 0. Pre-flight: reachability + the snapshot HARD GATE ─────────────────────
say "0. Pre-flight"
[ -d "${REPO}/agent-team/agent_team" ] || { err "no agent_team package under REPO=${REPO}"; exit 1; }
$SSH true || { err "cannot reach box (${BOX}); check the secrev alias / r720_seahaven key"; exit 1; }
echo "SNAPSHOT is Adam's step on the Hyper-V HOST (Claude's ssh reaches only the guest VM):"
echo " ssh Administrator@10.10.60.40 # PowerShell"
echo " Checkpoint-VM -Name sh-secrev -SnapshotName pre-deploy-${DATE}"
if [ "${ASSUME_SNAPSHOT:-0}" != "1" ]; then
confirm "Hyper-V snapshot taken and confirmed?" || { err "aborted: snapshot not confirmed"; exit 1; }
fi
# Record current health to compare after the restart.
read -r NRESTARTS_BEFORE < <($SSH "systemctl show ${SVC} -p NRestarts --value")
echo "current NRestarts=${NRESTARTS_BEFORE}, ActiveState=$($SSH "systemctl show ${SVC} -p ActiveState --value")"
# ── 1. Back up the ledger BEFORE any change ──────────────────────────────────
say "1. Back up the ledger"
$SSH "cp ~/orchestrator/agent-team/state/agent_team.sqlite ~/agent_team.sqlite.bak-${DATE}"
echo "ledger -> ~/agent_team.sqlite.bak-${DATE}"
# ── 2. Sync code the SAFE way: WHOLE package, never individual files ─────────
say "2. rsync the WHOLE agent_team package (structure preserved) + run-team.py"
rsync -az --exclude '__pycache__' --exclude '*.pyc' --exclude 'state' \
"${REPO}/agent-team/agent_team/" "${BOX}:orchestrator/agent-team/agent_team/"
rsync -az "${REPO}/agent-team/run-team.py" "${BOX}:orchestrator/agent-team/run-team.py"
if [ "${SYNC_DEPS:-0}" = "1" ]; then
say "2a. requirements changed -> rsync + reinstall into the box venv"
rsync -az "${REPO}/requirements.txt" "${BOX}:orchestrator/requirements.txt"
$SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && pip install --upgrade -r ~/orchestrator/requirements.txt'
fi
if [ "${SYNC_HANDBOOK:-0}" = "1" ]; then
say "2b. sync engineering handbook (WS5 context_provider source)"
if [ -d "${HANDBOOK_LOCAL}" ]; then
$SSH "mkdir -p ${HANDBOOK_REMOTE}"
rsync -az --delete --exclude .git "${HANDBOOK_LOCAL}/" "${BOX}:${HANDBOOK_REMOTE}/"
else
err "HANDBOOK_LOCAL=${HANDBOOK_LOCAL} not found; skipping (context_provider fail-safes to '')."
fi
fi
# ── 2c. Pre-restart import sanity (catch the crash before it loops) ──────────
say "2c. Pre-restart import sanity"
$SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && \
python3 -c "import agent_team.coordinator, agent_team.graph; from agent_team.nodes import planner; print(\"imports OK\")"' \
|| { err "import check FAILED — the sync is broken (likely a misplaced module). Fix the sync; do NOT restart."; exit 1; }
# ── 3. Restart ───────────────────────────────────────────────────────────────
say "3. Restart ${SVC}"
confirm "Restart the LIVE coordinator now?" || { err "skipped restart"; exit 0; }
$SSH "sudo systemctl restart ${SVC}"
if [ "${RESTART_STATUS:-0}" = "1" ]; then
$SSH 'sudo systemctl restart agent-team-status.service'
fi
# ── 4. VERIFY (mandatory) ────────────────────────────────────────────────────
say "4. Verify (is-active / NRestarts / threads / journal)"
sleep 3
STATE=$($SSH "systemctl is-active ${SVC}" || true)
read -r NRESTARTS_AFTER < <($SSH "systemctl show ${SVC} -p NRestarts --value")
MAINPID=$($SSH "systemctl show ${SVC} -p MainPID --value")
THREADS=$($SSH "ls /proc/${MAINPID}/task 2>/dev/null | wc -l" || echo 0)
echo "is-active=${STATE} NRestarts ${NRESTARTS_BEFORE} -> ${NRESTARTS_AFTER} MainPID=${MAINPID} threads=${THREADS}"
OK=1
[ "${STATE}" = "active" ] || { err "is-active=${STATE} (expected active; 'activating' = crash-loop)"; OK=0; }
[ "${NRESTARTS_AFTER}" -le "${NRESTARTS_BEFORE}" ] || { err "NRestarts climbed (${NRESTARTS_BEFORE}->${NRESTARTS_AFTER}) = crash-loop"; OK=0; }
[ "${THREADS}" -ge 5 ] || { err "thread count ${THREADS} (~6 expected; Socket Mode listener may be down)"; OK=0; }
if $SSH "journalctl -u ${SVC} --since '30 seconds ago' --no-pager" | grep -Eiq 'traceback|error'; then
err "journal shows Traceback/error in the last 30s"; OK=0
fi
if [ "${OK}" != "1" ]; then
err "DEPLOY VERIFY FAILED — rolling back is required."
echo "Evidence:"; $SSH "journalctl -u ${SVC} -n 80 --no-pager" || true
cat <<ROLLBACK_MSG
ROLLBACK (Adam, on the Hyper-V host 10.10.60.40):
Restore-VMSnapshot -Name pre-deploy-${DATE} -VMName sh-secrev -Confirm:\$false
The ledger backup ~/agent_team.sqlite.bak-${DATE} is the secondary safety net.
Re-verify health after rollback before doing anything else.
ROLLBACK_MSG
exit 1
fi
say "VERIFY PASSED — coordinator active, no restart climb, listener up."
echo "Reminder: merges to main remain held (deploy-before-merge)."
say "Done."