#!/usr/bin/env bash # deploy-r720.sh — SAFE, idempotent UPDATE of the live R720 agent-team coordinator. # # This is the GENERAL deploy path for ongoing agent-team code changes (the # worked, one-off example is deploy-r720-ws-rollout.sh). It is deploy-before-merge: # rsync the WHOLE agent_team package from your local checkout, restart the # coordinator, and VERIFY. Run it from the MAC. # # It exists because of two self-inflicted live crash-loops: # 1) Per-file rsync drops the subdir and lands e.g. nodes/planner.py at the # package root -> ImportError/TypeError crash-loop. This script rsyncs the # WHOLE agent_team/ tree (structure preserved), never individual files. # 2) A bad deploy reads as `activating` (systemd auto-restart), NOT a clean # error, and the daemon logs WARNING+ only (empty journal != healthy). So # this script VERIFIES is-active==active + NRestarts-didn't-climb + thread # count ~6 after the restart, and tells you to roll back via the snapshot # on failure. # # The Hyper-V snapshot is Adam's step on the WINDOWS HYPERVISOR (10.10.60.40) — # Claude's `ssh secrev` reaches only the guest VM. This script PROMPTS to # confirm the snapshot and refuses to proceed without it. # # It is idempotent and FAILS LOUDLY (set -euo pipefail). # # Usage: # REPO=~/Documents/repositories/orchestrator bash agent-team/scripts/deploy-r720.sh # Options (env): # SYNC_DEPS=1 also rsync requirements.txt + pip install into the box venv # (set this whenever requirements.txt changed — esp. the # langchain-* model stack, or models.py fails to import and # the non-Claude review/scan/build silently never run). # SYNC_HANDBOOK=1 also rsync the engineering handbook (WS5 context_provider). # RESTART_STATUS=1 also restart agent-team-status.service (the LAN dashboard). # ASSUME_SNAPSHOT=1 skip the interactive snapshot prompt (you confirmed it # out of band). Do NOT set this casually. set -euo pipefail # ── Config (override via env) ──────────────────────────────────────────────── BOX="${BOX:-secrev}" # ssh alias -> guest VM (key r720_seahaven, NOPASSWD sudo) REPO="${REPO:-$HOME/Documents/repositories/orchestrator}" HANDBOOK_LOCAL="${HANDBOOK_LOCAL:-$HOME/Documents/repositories/engineering-handbook}" HANDBOOK_REMOTE="${HANDBOOK_REMOTE:-.sea-haven/engineering-handbook}" # matches SEA_HAVEN_HANDBOOK_DIR DATE="${DATE:-$(date +%Y%m%d)}" SVC="agent-team-coordinator.service" SSH="ssh ${BOX}" say() { printf '\n\033[1;36m== %s\033[0m\n' "$*"; } err() { printf '\033[1;31m!! %s\033[0m\n' "$*" >&2; } confirm() { read -r -p "$1 [y/N] " a; [ "$a" = "y" ] || [ "$a" = "Y" ]; } # ── 0. Pre-flight: reachability + the snapshot HARD GATE ───────────────────── say "0. Pre-flight" [ -d "${REPO}/agent-team/agent_team" ] || { err "no agent_team package under REPO=${REPO}"; exit 1; } $SSH true || { err "cannot reach box (${BOX}); check the secrev alias / r720_seahaven key"; exit 1; } echo "SNAPSHOT is Adam's step on the Hyper-V HOST (Claude's ssh reaches only the guest VM):" echo " ssh Administrator@10.10.60.40 # PowerShell" echo " Checkpoint-VM -Name sh-secrev -SnapshotName pre-deploy-${DATE}" if [ "${ASSUME_SNAPSHOT:-0}" != "1" ]; then confirm "Hyper-V snapshot taken and confirmed?" || { err "aborted: snapshot not confirmed"; exit 1; } fi # Record current health to compare after the restart. read -r NRESTARTS_BEFORE < <($SSH "systemctl show ${SVC} -p NRestarts --value") echo "current NRestarts=${NRESTARTS_BEFORE}, ActiveState=$($SSH "systemctl show ${SVC} -p ActiveState --value")" # ── 1. Back up the ledger BEFORE any change ────────────────────────────────── say "1. Back up the ledger" $SSH "cp ~/orchestrator/agent-team/state/agent_team.sqlite ~/agent_team.sqlite.bak-${DATE}" echo "ledger -> ~/agent_team.sqlite.bak-${DATE}" # ── 2. Sync code the SAFE way: WHOLE package, never individual files ───────── say "2. rsync the WHOLE agent_team package (structure preserved) + run-team.py" rsync -az --exclude '__pycache__' --exclude '*.pyc' --exclude 'state' \ "${REPO}/agent-team/agent_team/" "${BOX}:orchestrator/agent-team/agent_team/" rsync -az "${REPO}/agent-team/run-team.py" "${BOX}:orchestrator/agent-team/run-team.py" if [ "${SYNC_DEPS:-0}" = "1" ]; then say "2a. requirements changed -> rsync + reinstall into the box venv" rsync -az "${REPO}/requirements.txt" "${BOX}:orchestrator/requirements.txt" $SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && pip install --upgrade -r ~/orchestrator/requirements.txt' fi if [ "${SYNC_HANDBOOK:-0}" = "1" ]; then say "2b. sync engineering handbook (WS5 context_provider source)" if [ -d "${HANDBOOK_LOCAL}" ]; then $SSH "mkdir -p ${HANDBOOK_REMOTE}" rsync -az --delete --exclude .git "${HANDBOOK_LOCAL}/" "${BOX}:${HANDBOOK_REMOTE}/" else err "HANDBOOK_LOCAL=${HANDBOOK_LOCAL} not found; skipping (context_provider fail-safes to '')." fi fi # ── 2c. Pre-restart import sanity (catch the crash before it loops) ────────── say "2c. Pre-restart import sanity" $SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && \ python3 -c "import agent_team.coordinator, agent_team.graph; from agent_team.nodes import planner; print(\"imports OK\")"' \ || { err "import check FAILED — the sync is broken (likely a misplaced module). Fix the sync; do NOT restart."; exit 1; } # ── 3. Restart ─────────────────────────────────────────────────────────────── say "3. Restart ${SVC}" confirm "Restart the LIVE coordinator now?" || { err "skipped restart"; exit 0; } $SSH "sudo systemctl restart ${SVC}" if [ "${RESTART_STATUS:-0}" = "1" ]; then $SSH 'sudo systemctl restart agent-team-status.service' fi # ── 4. VERIFY (mandatory) ──────────────────────────────────────────────────── say "4. Verify (is-active / NRestarts / threads / journal)" sleep 3 STATE=$($SSH "systemctl is-active ${SVC}" || true) read -r NRESTARTS_AFTER < <($SSH "systemctl show ${SVC} -p NRestarts --value") MAINPID=$($SSH "systemctl show ${SVC} -p MainPID --value") THREADS=$($SSH "ls /proc/${MAINPID}/task 2>/dev/null | wc -l" || echo 0) echo "is-active=${STATE} NRestarts ${NRESTARTS_BEFORE} -> ${NRESTARTS_AFTER} MainPID=${MAINPID} threads=${THREADS}" OK=1 [ "${STATE}" = "active" ] || { err "is-active=${STATE} (expected active; 'activating' = crash-loop)"; OK=0; } [ "${NRESTARTS_AFTER}" -le "${NRESTARTS_BEFORE}" ] || { err "NRestarts climbed (${NRESTARTS_BEFORE}->${NRESTARTS_AFTER}) = crash-loop"; OK=0; } [ "${THREADS}" -ge 5 ] || { err "thread count ${THREADS} (~6 expected; Socket Mode listener may be down)"; OK=0; } if $SSH "journalctl -u ${SVC} --since '30 seconds ago' --no-pager" | grep -Eiq 'traceback|error'; then err "journal shows Traceback/error in the last 30s"; OK=0 fi if [ "${OK}" != "1" ]; then err "DEPLOY VERIFY FAILED — rolling back is required." echo "Evidence:"; $SSH "journalctl -u ${SVC} -n 80 --no-pager" || true cat <