diff --git a/agent-team/scripts/deploy-r720.sh b/agent-team/scripts/deploy-r720.sh deleted file mode 100755 index 6012b16..0000000 --- a/agent-team/scripts/deploy-r720.sh +++ /dev/null @@ -1,140 +0,0 @@ -#!/usr/bin/env bash -# deploy-r720.sh — SAFE, idempotent UPDATE of the live R720 agent-team coordinator. -# -# This is the GENERAL deploy path for ongoing agent-team code changes (the -# worked, one-off example is deploy-r720-ws-rollout.sh). It is deploy-before-merge: -# rsync the WHOLE agent_team package from your local checkout, restart the -# coordinator, and VERIFY. Run it from the MAC. -# -# It exists because of two self-inflicted live crash-loops: -# 1) Per-file rsync drops the subdir and lands e.g. nodes/planner.py at the -# package root -> ImportError/TypeError crash-loop. This script rsyncs the -# WHOLE agent_team/ tree (structure preserved), never individual files. -# 2) A bad deploy reads as `activating` (systemd auto-restart), NOT a clean -# error, and the daemon logs WARNING+ only (empty journal != healthy). So -# this script VERIFIES is-active==active + NRestarts-didn't-climb + thread -# count ~6 after the restart, and tells you to roll back via the snapshot -# on failure. -# -# The Hyper-V snapshot is Adam's step on the WINDOWS HYPERVISOR (10.10.60.40) — -# Claude's `ssh secrev` reaches only the guest VM. This script PROMPTS to -# confirm the snapshot and refuses to proceed without it. -# -# It is idempotent and FAILS LOUDLY (set -euo pipefail). -# -# Usage: -# REPO=~/Documents/repositories/orchestrator bash agent-team/scripts/deploy-r720.sh -# Options (env): -# SYNC_DEPS=1 also rsync requirements.txt + pip install into the box venv -# (set this whenever requirements.txt changed — esp. the -# langchain-* model stack, or models.py fails to import and -# the non-Claude review/scan/build silently never run). -# SYNC_HANDBOOK=1 also rsync the engineering handbook (WS5 context_provider). -# RESTART_STATUS=1 also restart agent-team-status.service (the LAN dashboard). -# ASSUME_SNAPSHOT=1 skip the interactive snapshot prompt (you confirmed it -# out of band). Do NOT set this casually. -set -euo pipefail - -# ── Config (override via env) ──────────────────────────────────────────────── -BOX="${BOX:-secrev}" # ssh alias -> guest VM (key r720_seahaven, NOPASSWD sudo) -REPO="${REPO:-$HOME/Documents/repositories/orchestrator}" -HANDBOOK_LOCAL="${HANDBOOK_LOCAL:-$HOME/Documents/repositories/engineering-handbook}" -HANDBOOK_REMOTE="${HANDBOOK_REMOTE:-.sea-haven/engineering-handbook}" # matches SEA_HAVEN_HANDBOOK_DIR -DATE="${DATE:-$(date +%Y%m%d)}" -SVC="agent-team-coordinator.service" -SSH="ssh ${BOX}" - -say() { printf '\n\033[1;36m== %s\033[0m\n' "$*"; } -err() { printf '\033[1;31m!! %s\033[0m\n' "$*" >&2; } -confirm() { read -r -p "$1 [y/N] " a; [ "$a" = "y" ] || [ "$a" = "Y" ]; } - -# ── 0. Pre-flight: reachability + the snapshot HARD GATE ───────────────────── -say "0. Pre-flight" -[ -d "${REPO}/agent-team/agent_team" ] || { err "no agent_team package under REPO=${REPO}"; exit 1; } -$SSH true || { err "cannot reach box (${BOX}); check the secrev alias / r720_seahaven key"; exit 1; } - -echo "SNAPSHOT is Adam's step on the Hyper-V HOST (Claude's ssh reaches only the guest VM):" -echo " ssh Administrator@10.10.60.40 # PowerShell" -echo " Checkpoint-VM -Name sh-secrev -SnapshotName pre-deploy-${DATE}" -if [ "${ASSUME_SNAPSHOT:-0}" != "1" ]; then - confirm "Hyper-V snapshot taken and confirmed?" || { err "aborted: snapshot not confirmed"; exit 1; } -fi - -# Record current health to compare after the restart. -read -r NRESTARTS_BEFORE < <($SSH "systemctl show ${SVC} -p NRestarts --value") -echo "current NRestarts=${NRESTARTS_BEFORE}, ActiveState=$($SSH "systemctl show ${SVC} -p ActiveState --value")" - -# ── 1. Back up the ledger BEFORE any change ────────────────────────────────── -say "1. Back up the ledger" -$SSH "cp ~/orchestrator/agent-team/state/agent_team.sqlite ~/agent_team.sqlite.bak-${DATE}" -echo "ledger -> ~/agent_team.sqlite.bak-${DATE}" - -# ── 2. Sync code the SAFE way: WHOLE package, never individual files ───────── -say "2. rsync the WHOLE agent_team package (structure preserved) + run-team.py" -rsync -az --exclude '__pycache__' --exclude '*.pyc' --exclude 'state' \ - "${REPO}/agent-team/agent_team/" "${BOX}:orchestrator/agent-team/agent_team/" -rsync -az "${REPO}/agent-team/run-team.py" "${BOX}:orchestrator/agent-team/run-team.py" - -if [ "${SYNC_DEPS:-0}" = "1" ]; then - say "2a. requirements changed -> rsync + reinstall into the box venv" - rsync -az "${REPO}/requirements.txt" "${BOX}:orchestrator/requirements.txt" - $SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && pip install --upgrade -r ~/orchestrator/requirements.txt' -fi - -if [ "${SYNC_HANDBOOK:-0}" = "1" ]; then - say "2b. sync engineering handbook (WS5 context_provider source)" - if [ -d "${HANDBOOK_LOCAL}" ]; then - $SSH "mkdir -p ${HANDBOOK_REMOTE}" - rsync -az --delete --exclude .git "${HANDBOOK_LOCAL}/" "${BOX}:${HANDBOOK_REMOTE}/" - else - err "HANDBOOK_LOCAL=${HANDBOOK_LOCAL} not found; skipping (context_provider fail-safes to '')." - fi -fi - -# ── 2c. Pre-restart import sanity (catch the crash before it loops) ────────── -say "2c. Pre-restart import sanity" -$SSH 'cd ~/orchestrator/agent-team && . .venv/bin/activate && \ - python3 -c "import agent_team.coordinator, agent_team.graph; from agent_team.nodes import planner; print(\"imports OK\")"' \ - || { err "import check FAILED — the sync is broken (likely a misplaced module). Fix the sync; do NOT restart."; exit 1; } - -# ── 3. Restart ─────────────────────────────────────────────────────────────── -say "3. Restart ${SVC}" -confirm "Restart the LIVE coordinator now?" || { err "skipped restart"; exit 0; } -$SSH "sudo systemctl restart ${SVC}" -if [ "${RESTART_STATUS:-0}" = "1" ]; then - $SSH 'sudo systemctl restart agent-team-status.service' -fi - -# ── 4. VERIFY (mandatory) ──────────────────────────────────────────────────── -say "4. Verify (is-active / NRestarts / threads / journal)" -sleep 3 -STATE=$($SSH "systemctl is-active ${SVC}" || true) -read -r NRESTARTS_AFTER < <($SSH "systemctl show ${SVC} -p NRestarts --value") -MAINPID=$($SSH "systemctl show ${SVC} -p MainPID --value") -THREADS=$($SSH "ls /proc/${MAINPID}/task 2>/dev/null | wc -l" || echo 0) -echo "is-active=${STATE} NRestarts ${NRESTARTS_BEFORE} -> ${NRESTARTS_AFTER} MainPID=${MAINPID} threads=${THREADS}" - -OK=1 -[ "${STATE}" = "active" ] || { err "is-active=${STATE} (expected active; 'activating' = crash-loop)"; OK=0; } -[ "${NRESTARTS_AFTER}" -le "${NRESTARTS_BEFORE}" ] || { err "NRestarts climbed (${NRESTARTS_BEFORE}->${NRESTARTS_AFTER}) = crash-loop"; OK=0; } -[ "${THREADS}" -ge 5 ] || { err "thread count ${THREADS} (~6 expected; Socket Mode listener may be down)"; OK=0; } -if $SSH "journalctl -u ${SVC} --since '30 seconds ago' --no-pager" | grep -Eiq 'traceback|error'; then - err "journal shows Traceback/error in the last 30s"; OK=0 -fi - -if [ "${OK}" != "1" ]; then - err "DEPLOY VERIFY FAILED — rolling back is required." - echo "Evidence:"; $SSH "journalctl -u ${SVC} -n 80 --no-pager" || true - cat <&2; } if command -v cfn-lint >/dev/null; then # Prune generated/vendored trees (cdk.out, node_modules, …): scanning synthesized output is # wrong and, on CDK repos, explodes the arg list / stalls the scanners. - TPLS="$(scope_paths | xargs -I{} find {} \( -type d \( -name cdk.out -o -name node_modules -o -name .git -o -name .claude -o -name .aws-sam -o -name .venv -o -name venv -o -name dist -o -name build \) -prune \) -o \( -type f \( -name '*.yaml' -o -name '*.yml' \) -print \) 2>/dev/null \ - | xargs -I{} sh -c 'grep -lE "AWSTemplateFormatVersion|Transform: *AWS::Serverless" "{}" 2>/dev/null || true')" + # `find -print0 | xargs -0 grep -lE` (was `… | xargs -I{} sh -c 'grep -l "{}"'`): the grep stage + # is the one that overflows. `xargs -I{}` packs every matched path — long, absolute, deep-worktree + # paths on this monorepo — into one assembled command and dies with "command line cannot be + # assembled, too long", emitting zero findings and blocking the push. NUL-delimited `xargs -0 grep` + # splits across invocations transparently (batches by ARG_MAX, never overflows), is safe for paths + # with spaces/newlines, and `grep -l` reports the same matching files as the old per-file grep. + # The first `xargs -I{} find {}` keeps the start path first (find needs it before the expression) + # and is bounded by the scope-path count, so it is not an overflow risk. `--no-run-if-empty` is + # GNU-only, so the trailing `|| true` absorbs grep's exit 1 on no-match / no-files, matching the + # old per-file `… || true` so TPLS is "list of templates, or empty" and never fails the gate. + TPLS="$(scope_paths | xargs -I{} find {} \( -type d \( -name cdk.out -o -name node_modules -o -name .git -o -name .claude -o -name .aws-sam -o -name .venv -o -name venv -o -name dist -o -name build \) -prune \) -o \( -type f \( -name '*.yaml' -o -name '*.yml' \) -print0 \) 2>/dev/null \ + | xargs -0 grep -lE "AWSTemplateFormatVersion|Transform: *AWS::Serverless" 2>/dev/null || true)" if [ -n "$TPLS" ]; then # shellcheck disable=SC2086 RAW="$(cfn-lint -f json $TPLS 2>/dev/null || true)"