From 18412c748207f0e6158f9cd39d220045b9bdef3a Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 14:59:42 -0400 Subject: [PATCH 01/16] Remove parked security-review CI drafts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CI was swapped for the global git hooks + nightly VM sweep (solo dev), so the parked ci/*.yml and CI-BACKSTOP-NOTES.md were dead weight — a defective workflow in-tree is a foot-gun. Recover from history if the team grows. --- security-review/ci/security-review.yml | 61 -------------------------- 1 file changed, 61 deletions(-) delete mode 100644 security-review/ci/security-review.yml diff --git a/security-review/ci/security-review.yml b/security-review/ci/security-review.yml deleted file mode 100644 index d52d930..0000000 --- a/security-review/ci/security-review.yml +++ /dev/null @@ -1,61 +0,0 @@ -# Sea Haven security-review CI backstop (Phase 3). -# Drop into a target repo as .github/workflows/security-review.yml, OR (preferred, per -# engineering-handbook/cicd.md) promote into Sea-Haven-Industries/.github as a reusable workflow. -# -# This is the UNBYPASSABLE deterministic backstop: local hooks can be skipped with --no-verify, -# this cannot. It runs the SAME review.sh as local. The agentic detector/verifier pass (Path B) -# is gated behind SECURITY_REVIEW_AGENTIC=1 and requires the headless orchestrator runner + -# ANTHROPIC creds (see DEPLOY-R720.md) — until that exists, CI runs the scanners-only gate. -name: security-review -on: - pull_request: - workflow_dispatch: - -permissions: - contents: read - -jobs: - security-review: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - - name: Install scanners - run: | - python3 -m pip install --quiet pipx && python3 -m pipx ensurepath - pipx install semgrep >/dev/null - pipx install checkov >/dev/null - pipx install pip-audit >/dev/null - pip install --quiet cfn-lint - # gitleaks binary - curl -sSL https://github.com/gitleaks/gitleaks/releases/download/v8.30.1/gitleaks_8.30.1_linux_x64.tar.gz \ - | tar -xz -C /usr/local/bin gitleaks - echo "$HOME/.local/bin" >> "$GITHUB_PATH" - - - name: Fetch review.sh - run: | - # Pin to the orchestrator repo / a release artifact. Placeholder: vendor a copy or curl a tag. - git clone --depth 1 https://github.com/amoussa1229/orchestrator /tmp/orch - chmod +x /tmp/orch/security-review/review.sh - - - name: Run gate (scanners; agentic if enabled) - env: - SECURITY_REVIEW_AGENTIC: ${{ vars.SECURITY_REVIEW_AGENTIC }} # set to 1 once headless runner exists - ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - run: | - SCOPE="${SECURITY_REVIEW_SCOPE:-src scripts template.yaml}" - if [ "${SECURITY_REVIEW_AGENTIC:-0}" = "1" ]; then - python3 /tmp/orch/security-review/run_headless.py --scope "$SCOPE" --out /tmp/agent.json . - /tmp/orch/security-review/review.sh --scope "$SCOPE" --agent-findings /tmp/agent.json \ - --suppressions .security-review/suppressions.json --json-out /tmp/review.json . - else - /tmp/orch/security-review/review.sh --scope "$SCOPE" --scanners-only \ - --suppressions .security-review/suppressions.json --json-out /tmp/review.json . - fi - - - name: Upload findings - if: always() - uses: actions/upload-artifact@v4 - with: - name: security-review-findings - path: /tmp/review.json -- 2.50.1 From a3ab3f5f40d226301ac1929518338df0fdc92cea Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:00:00 -0400 Subject: [PATCH 02/16] Make security-review hooks and skill installable from the repo MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The global pre-push hook, the /sh-security-review prompt, and finding.schema.json previously lived only in ~/.config/git and ~/.claude (untracked) — unreproducible. Source them here: add hooks/pre-push, rewrite install-hooks.sh with a --global mode (lays down both hooks, sets core.hooksPath, links skill+schema into ~/.claude) and a per-repo mode. Align pre-commit with pre-push (honor skip marker + suppressions). Add semgrep p/javascript so the scanners cover the org's Node/.NET repos. --- security-review/finding.schema.json | 69 +++++++++++++++ security-review/hooks/pre-commit | 13 ++- security-review/hooks/pre-push | 26 ++++++ security-review/install-hooks.sh | 91 +++++++++++++++++--- security-review/review.sh | 19 +++-- security-review/skill/sh-security-review.md | 93 +++++++++++++++++++++ 6 files changed, 292 insertions(+), 19 deletions(-) create mode 100644 security-review/finding.schema.json create mode 100755 security-review/hooks/pre-push create mode 100644 security-review/skill/sh-security-review.md diff --git a/security-review/finding.schema.json b/security-review/finding.schema.json new file mode 100644 index 0000000..52be56e --- /dev/null +++ b/security-review/finding.schema.json @@ -0,0 +1,69 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "title": "Sea Haven security-review finding", + "description": "Structured finding contract for /sh-security-review. Same shape for interactive (Path A) and automated (Path B) runs, and the input review.sh reads to make the block decision.", + "type": "object", + "required": ["findings", "summary"], + "properties": { + "findings": { + "type": "array", + "items": { + "type": "object", + "required": ["id", "title", "severity", "cwe", "file", "category", "data_flow", "proof", "status"], + "properties": { + "id": { "type": "string", "description": "stable slug, e.g. sqli-payment-handler-get-payment" }, + "title": { "type": "string" }, + "severity": { + "type": "string", + "enum": ["critical", "high", "medium", "low", "info", "unverified"], + "description": "unverified = a claim with no accepted proof; auto-downgraded from its claimed severity" + }, + "claimed_severity": { + "type": "string", + "enum": ["critical", "high", "medium", "low", "info"], + "description": "the detector's original severity before the verifier ruled" + }, + "cwe": { "type": "string", "pattern": "^CWE-[0-9]+$" }, + "file": { "type": "string" }, + "line": { "type": ["integer", "null"] }, + "category": { + "type": "string", + "enum": ["injection", "authz", "secrets-crypto", "iac-iam", "web-client", "logic", "other"] + }, + "data_flow": { + "type": "string", + "description": "numbered plain-English trace from untrusted source to dangerous sink" + }, + "proof": { + "type": "object", + "required": ["input", "outcome"], + "properties": { + "input": { "type": "string", "description": "concrete malicious input / trigger" }, + "outcome": { "type": "string", "description": "the specific bad result it produces" }, + "test": { "type": ["string", "null"], "description": "optional failing-test sketch" } + } + }, + "status": { + "type": "string", + "enum": ["confirmed", "unverified", "suppressed"], + "description": "confirmed = verifier accepted proof; unverified = no accepted proof; suppressed = dismissed with justification" + }, + "suppression_justification": { + "type": ["string", "null"], + "description": "REQUIRED when status=suppressed; logged and surfaced in the report" + }, + "recommendation": { "type": "string" } + } + } + }, + "summary": { + "type": "object", + "required": ["confirmed_critical", "confirmed_high", "block"], + "properties": { + "confirmed_critical": { "type": "integer" }, + "confirmed_high": { "type": "integer" }, + "block": { "type": "boolean", "description": "true if any confirmed critical/high is unsuppressed (the gate condition)" } + } + } + } +} diff --git a/security-review/hooks/pre-commit b/security-review/hooks/pre-commit index 242c568..c28b1f6 100755 --- a/security-review/hooks/pre-commit +++ b/security-review/hooks/pre-commit @@ -1,9 +1,11 @@ #!/usr/bin/env bash # Sea Haven security-review pre-commit hook: FAST deterministic scanners only (sub-30s). # The full agentic review is the on-demand /sh-security-review slash command — run that before pushing. -# CI re-runs review.sh as the unbypassable backstop, so --no-verify here only skips local fast feedback. -set -euo pipefail -REPO_ROOT="$(git rev-parse --show-toplevel)" +# Honors the same skip/suppress controls as pre-push so a suppressed FP doesn't block the commit. +# --no-verify skips this local fast feedback; the pre-push hook + nightly VM sweep are the backstop. +set -uo pipefail +REPO_ROOT="$(git rev-parse --show-toplevel 2>/dev/null)" || exit 0 +[ -f "$REPO_ROOT/.security-review-skip" ] && exit 0 REVIEW_SH="${SH_REVIEW_SH:-$HOME/Documents/repositories/orchestrator/security-review/review.sh}" if [ ! -f "$REVIEW_SH" ]; then echo "security-review: review.sh not found at $REVIEW_SH (set SH_REVIEW_SH to override) — skipping" >&2 @@ -11,4 +13,7 @@ if [ ! -f "$REVIEW_SH" ]; then fi # Nothing staged -> nothing to do. git diff --cached --name-only --diff-filter=ACM | grep -q . || exit 0 -bash "$REVIEW_SH" --scanners-only "$REPO_ROOT" +SUP=() +[ -f "$REPO_ROOT/.security-review/suppressions.json" ] && SUP=(--suppressions "$REPO_ROOT/.security-review/suppressions.json") +# ${SUP[@]+"${SUP[@]}"} = bash-3.2-safe expansion of a possibly-empty array under set -u. +exec bash "$REVIEW_SH" --scanners-only ${SUP[@]+"${SUP[@]}"} "$REPO_ROOT" diff --git a/security-review/hooks/pre-push b/security-review/hooks/pre-push new file mode 100755 index 0000000..74118d8 --- /dev/null +++ b/security-review/hooks/pre-push @@ -0,0 +1,26 @@ +#!/usr/bin/env bash +# Sea Haven global pre-push security gate — fast deterministic scanners (review.sh --scanners-only). +# Installed via install-hooks.sh --global: lays this down at ~/.config/git/hooks/pre-push and sets +# git config --global core.hooksPath ~/.config/git/hooks +# Skip a repo: add a .security-review-skip file at its root. Bypass once: git push --no-verify. +# Deep agentic pass = on-demand /sh-security-review; nightly VM sweep = the backstop. +set -uo pipefail +REPO_ROOT="$(git rev-parse --show-toplevel 2>/dev/null)" || exit 0 +[ -f "$REPO_ROOT/.security-review-skip" ] && exit 0 +REVIEW_SH="${SH_REVIEW_SH:-$HOME/Documents/repositories/orchestrator/security-review/review.sh}" +if [ -f "$REVIEW_SH" ]; then + SUP=() + [ -f "$REPO_ROOT/.security-review/suppressions.json" ] && SUP=(--suppressions "$REPO_ROOT/.security-review/suppressions.json") + echo "security-review: scanning $REPO_ROOT (scanners-only) before push..." >&2 + # ${SUP[@]+"${SUP[@]}"} = bash-3.2-safe expansion of a possibly-empty array under set -u. + if ! bash "$REVIEW_SH" --scanners-only ${SUP[@]+"${SUP[@]}"} "$REPO_ROOT"; then + echo "security-review: BLOCKED (confirmed crit/high). Fix it, suppress with justification, or 'git push --no-verify' to override." >&2 + exit 1 + fi +else + echo "security-review: review.sh not found at $REVIEW_SH (set SH_REVIEW_SH) — skipping gate" >&2 +fi +# Don't silently disable a repo-local pre-push hook: chain to it if present. +LOCAL_HOOK="$REPO_ROOT/.git/hooks/pre-push" +[ -x "$LOCAL_HOOK" ] && exec "$LOCAL_HOOK" "$@" +exit 0 diff --git a/security-review/install-hooks.sh b/security-review/install-hooks.sh index 87b7668..6a10a78 100755 --- a/security-review/install-hooks.sh +++ b/security-review/install-hooks.sh @@ -1,12 +1,83 @@ #!/usr/bin/env bash -# Install the Sea Haven security-review pre-commit hook into a target repo. -# Usage: install-hooks.sh /path/to/repo +# install-hooks.sh — install the Sea Haven security-review git hooks + skill assets. +# +# Two modes: +# install-hooks.sh --global Lay the hooks down once for EVERY repo on this machine: +# writes ~/.config/git/hooks/{pre-commit,pre-push}, sets +# git config --global core.hooksPath, and links the skill +# prompt + finding.schema.json into ~/.claude (Path A). +# install-hooks.sh /path/to/repo Per-repo install: copy the hooks into /.git/hooks +# (use when a repo sets its own local core.hooksPath, e.g. +# husky, which would otherwise shadow the global hook). +# install-hooks.sh --help +# +# The hooks run review.sh --scanners-only (fast, deterministic). The full agentic review is the +# on-demand /sh-security-review skill; the nightly VM sweep is the backstop. Idempotent + re-runnable. set -euo pipefail -REPO="${1:?usage: install-hooks.sh /path/to/repo}" -SRC="$(cd "$(dirname "$0")" && pwd)/hooks/pre-commit" -[ -d "$REPO/.git" ] || { echo "not a git repo: $REPO" >&2; exit 1; } -HOOK="$REPO/.git/hooks/pre-commit" -if [ -f "$HOOK" ]; then echo "warning: existing pre-commit hook at $HOOK will be overwritten" >&2; fi -cp "$SRC" "$HOOK"; chmod +x "$HOOK" -echo "installed security-review pre-commit hook -> $HOOK" -echo "note: hook runs deterministic scanners only; full review is /sh-security-review (on demand)." + +SRC="$(cd "$(dirname "$0")" && pwd)" +GLOBAL_HOOKS="$HOME/.config/git/hooks" +CLAUDE_DIR="$HOME/.claude" + +usage() { grep '^#' "$0" | sed 's/^# \{0,1\}//'; } + +install_one() { # install_one + local src="$1" dest="$2" + cp "$src" "$dest" + chmod +x "$dest" +} + +link_asset() { # link_asset (symlink so the repo stays source of truth) + local src="$1" dest="$2" + mkdir -p "$(dirname "$dest")" + ln -sfn "$src" "$dest" + echo " linked $dest -> $src" +} + +case "${1:-}" in + -h|--help|"") usage; exit 0;; + + --global) + echo "== Installing Sea Haven security-review hooks globally ==" + mkdir -p "$GLOBAL_HOOKS" + + # Warn (don't clobber silently) if a different global hooksPath is already set. + CURRENT="$(git config --global --get core.hooksPath || true)" + if [ -n "$CURRENT" ] && [ "$CURRENT" != "$GLOBAL_HOOKS" ]; then + echo " WARNING: git config --global core.hooksPath is already '$CURRENT'." >&2 + echo " Overwriting it with '$GLOBAL_HOOKS'. Re-point manually if that was intentional." >&2 + fi + + install_one "$SRC/hooks/pre-commit" "$GLOBAL_HOOKS/pre-commit" + install_one "$SRC/hooks/pre-push" "$GLOBAL_HOOKS/pre-push" + git config --global core.hooksPath "$GLOBAL_HOOKS" + echo " installed pre-commit + pre-push -> $GLOBAL_HOOKS" + echo " set git config --global core.hooksPath = $GLOBAL_HOOKS" + + # Path A assets: link the on-demand skill prompt + finding schema into ~/.claude. + link_asset "$SRC/skill/sh-security-review.md" "$CLAUDE_DIR/commands/sh-security-review.md" + link_asset "$SRC/finding.schema.json" "$CLAUDE_DIR/security-review/finding.schema.json" + + echo + echo "Done. Every repo on this machine is now gated by review.sh --scanners-only before push." + echo "Caveats: a repo that sets its OWN local core.hooksPath (e.g. husky) overrides this global hook" + echo " — run 'install-hooks.sh ' to gate it per-repo. Skip a repo with a" + echo " .security-review-skip file at its root; bypass once with 'git push --no-verify'." + ;; + + --*) + echo "unknown option: $1" >&2; usage; exit 2;; + + *) + REPO="$1" + [ -d "$REPO/.git" ] || { echo "not a git repo: $REPO" >&2; exit 1; } + echo "== Installing security-review hooks into $REPO/.git/hooks ==" + for h in pre-commit pre-push; do + DEST="$REPO/.git/hooks/$h" + [ -f "$DEST" ] && echo " warning: existing $h hook at $DEST will be overwritten" >&2 + install_one "$SRC/hooks/$h" "$DEST" + echo " installed $h -> $DEST" + done + echo "note: hooks run deterministic scanners only; full review is /sh-security-review (on demand)." + ;; +esac diff --git a/security-review/review.sh b/security-review/review.sh index eb713f2..af5f766 100755 --- a/security-review/review.sh +++ b/security-review/review.sh @@ -67,7 +67,7 @@ SCOPE_PATHS="$(scope_paths)" # --- semgrep (SAST: injection/authz/xss/secrets) --- if command -v semgrep >/dev/null; then # shellcheck disable=SC2086 - if SG="$(semgrep --config p/security-audit --config p/secrets --json --metrics=off $SCOPE_PATHS 2>/dev/null)"; then + if SG="$(semgrep --config p/security-audit --config p/secrets --config p/javascript --json --metrics=off $SCOPE_PATHS 2>/dev/null)"; then NORM="$(echo "$SG" | jq '[.results[] | { id: ("semgrep-" + (.check_id|split(".")|last) + "-" + (.start.line|tostring)), title: ((.check_id|split(".")|last) + ": " + ((.extra.message // "")[0:120])), @@ -79,15 +79,24 @@ if command -v semgrep >/dev/null; then else echo " [semgrep] run failed" >&2; fi else note_missing semgrep "brew install semgrep"; fi -# --- gitleaks (hardcoded secrets) --- +# --- gitleaks (hardcoded secrets) — git-mode respects .gitignore (skips gitignored .env etc.) --- if command -v gitleaks >/dev/null; then GLALL="$TMP/gl.json"; echo '[]' > "$GLALL" - for p in $SCOPE_PATHS; do - gitleaks dir "$p" --report-format json --report-path "$TMP/gl1.json" >/dev/null 2>&1 || true + if git -C "$TARGET" rev-parse --is-inside-work-tree >/dev/null 2>&1; then + # Scan committed content at the repo root; gitignored files (e.g. a local .env with real + # keys) are excluded by design, so the gate never false-blocks on them. Same JSON schema. + gitleaks git "$TARGET" --report-format json --report-path "$TMP/gl1.json" >/dev/null 2>&1 || true if [ -s "$TMP/gl1.json" ]; then jq -s '.[0]+(.[1] // [])' "$GLALL" "$TMP/gl1.json" > "$GLALL.t" && mv "$GLALL.t" "$GLALL"; rm -f "$TMP/gl1.json" fi - done + else + for p in $SCOPE_PATHS; do + gitleaks dir "$p" --report-format json --report-path "$TMP/gl1.json" >/dev/null 2>&1 || true + if [ -s "$TMP/gl1.json" ]; then + jq -s '.[0]+(.[1] // [])' "$GLALL" "$TMP/gl1.json" > "$GLALL.t" && mv "$GLALL.t" "$GLALL"; rm -f "$TMP/gl1.json" + fi + done + fi NORM="$(jq '[.[] | { id: ("gitleaks-" + .RuleID + "-" + (.StartLine|tostring)), title: ("secret: " + .Description), severity: "high", cwe: "CWE-798", diff --git a/security-review/skill/sh-security-review.md b/security-review/skill/sh-security-review.md new file mode 100644 index 0000000..96dcc45 --- /dev/null +++ b/security-review/skill/sh-security-review.md @@ -0,0 +1,93 @@ +--- +name: sh-security-review +description: High-recall agentic security review with anti-complacency structure. Opus fans out N narrow fresh-context detectors over the target, a separate fresh-context verifier demands proof-of-exploit or downgrades to unverified, then findings are emitted in the structured schema with a block decision. Returns severity-ranked findings each carrying a concrete proof. +--- + +# Sea Haven Security Review (detector fan-out + proof-or-kill verifier) + +A high-recall security review built to resist reviewer complacency. Instead of one model judging the +whole surface (that's `sh-build-review`), this fans out **N narrow detectors, each fresh context and +adversarial**, then a **separate verifier** with the opposing incentive demands a concrete +proof-of-exploit for every candidate or downgrades it to `unverified`. Output is the structured +finding schema (`~/.claude/security-review/finding.schema.json`) plus a block decision. + +This is the interactive (Path A) entry point and runs under the Max subscription. The same prompts and +schema are reused by the automated `review.sh` (Path B) later. See memory `project-security-review-agent`. + +## When to Use +- Security review of a branch, working tree, file, or whole repo +- Before pushing payments/auth/IaC/input-handling changes +- As the high-recall pass; complements `/security-review`, `/code-review ultra`, and `sh-build-review` + +## Arguments +- Optional target: a path, file, branch, or diff. Default: the current working tree / branch diff vs main. +- Optional `--scope `: restrict the scan (e.g. `src/ infra/ web/`). + +## Mechanism (Opus runs these) + +Opus is the main loop. It scopes the target, runs the detector fan-out and verifier as subagents +(each with **fresh context** so no "we already passed 10 files" approval prior accumulates), then +applies the gate rule and reports. Opus does NOT soften the gate; the block condition is mechanical. + +### 1. Scope +Identify the files in scope (respect `--scope`; never scan an answer key or `.git`). Note the languages +present (Python/Lambda, .NET, React/JS, SAM/CDK IaC) so detectors apply the right checklist. + +### 2. Detector fan-out (parallel, fresh context, closed checklist, adversarial) +Spawn these detectors as **separate parallel subagents** (`subagent_type: general-purpose`). Each gets +ONLY its category, the schema, and the adversarial framing. Each must, per checklist item, either cite a +specific safe line OR file a finding — no open "looks fine" judgment. + +Detectors: +- **injection** — SQL/command/template injection, unsafe deserialization, SSRF, path/file traversal, XXE +- **authz** — broken object-level auth/IDOR, missing access checks, missing webhook/Slack signature verification, auth bypass +- **secrets-crypto** — hardcoded secrets/keys, weak/again crypto (MD5/SHA1/unsalted), sensitive data in logs/errors/responses, wrong SSM-vs-Secrets-Manager +- **iac-iam** — wildcard IAM actions/resources, public buckets/endpoints, open security-group ingress (0.0.0.0/0), missing encryption, over-broad trust policies +- **web-client** — XSS (incl. dangerouslySetInnerHTML), CSRF, open redirect, client-side secret exposure +- **logic** — broken multi-step invariants, race conditions, missing tenant isolation, auth-state confusion + +Detector prompt template (fill `{CATEGORY}`, `{CHECKLIST}`, `{SCOPE}`): +``` +You are a hostile {CATEGORY} security auditor for Sea Haven. Assume this code is hostile and the author +missed something. Audit ONLY {CATEGORY} issues in: {SCOPE}. Read the actual files. +For each checklist item, either name a specific line that is safe, OR file a finding. Do not hand-wave. +Checklist: {CHECKLIST} +Return a JSON array of findings, each: {id, title, claimed_severity (critical|high|medium|low|info), +cwe (CWE-####), file, line, category:"{CATEGORY}", data_flow (numbered source->sink trace), +proof:{input, outcome, test|null}, recommendation}. If none, return []. No prose outside the JSON. +``` + +### 3. Verifier (separate subagent, fresh context, proof-or-kill) +Spawn one or more **verifier** subagents with the OPPOSING incentive. The verifier did not find these and +is rewarded for killing weak claims. For each candidate it demands a concrete, plausible proof. +``` +You are a skeptical exploitation verifier. You did NOT find these; your job is to REFUTE weak claims. +For each candidate finding, decide: is there a concrete, plausible proof-of-exploit (a specific malicious +input and the specific bad outcome, consistent with the code)? +- If yes: status="confirmed", keep severity = claimed_severity, tighten the proof. +- If no / speculative / not reachable: status="unverified" and severity="unverified". Default to unverified + when uncertain. A confident assertion without a demonstrable input is NOT proof. +Return the findings array with status, severity, and the verified proof. No prose outside the JSON. +``` + +### 4. Merge + gate (mechanical, Opus does not soften) +- Dedup by (file, cwe, nearby line); keep the highest accepted severity. +- `summary.confirmed_critical` / `confirmed_high` = counts of status=confirmed at that severity. +- `summary.block = true` if any unsuppressed confirmed critical/high exists. +- A finding may be suppressed ONLY with a written `suppression_justification`, which is surfaced in the report. + No silent dismissal: a critical/high with no proof becomes `unverified` (still listed), never deleted. + +### 5. Report +Emit the schema JSON, then a human summary: BLOCK/PASS, confirmed findings highest-severity-first, each +with file:line, the numbered data-flow trace, and the proof. List unverified and suppressed separately so +nothing is silently dropped. The reader reviews proofs, not raw code. + +## Relationship to existing tooling +- `sh-build-review` — single deep Fable pass over a change surface (depth, one reasoner). +- `sh-security-review` — high-recall fan-out + proof-or-kill verifier (breadth + anti-complacency). +- IAM/policy and Lambda-signature changes still require the mandatory cross-family review per global instructions; this does not replace it. + +## Output +- Structured findings (schema) + block decision +- Confirmed findings with proofs, unverified and suppressed listed separately +- HARD STOP / BLOCK surfaced when `summary.block` is true -- 2.50.1 From 537e83975b6127abe7c6111043265a6600b9df71 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:00:00 -0400 Subject: [PATCH 03/16] Rewrite nightly sweep as two-tier clean-clone auto-discovery Replace the opt-in sweep-targets allowlist with zero-wiring discovery: enumerate org repos via the GitHub REST API (curl + read-only GH_TOKEN, no gh dependency) and mirror each as a shallow clean clone (git clone --depth=1, default branch from the API) into ~/repo-mirrors. Scanning server-side clones keeps local .env secrets out of scope. Tier 1 runs deterministic scanners over every repo nightly ($0 Claude); tier 2 runs the agentic pass over a budget-bounded round-robin rotation with a persistent cycle pointer, so the draw on the shared Max limits stays bounded and coverage never goes silently incomplete (COVERAGE ALARM if the rotation falls behind). Skip = committed marker or central list (marker-skips logged). Redact secrets from Slack; reports mode 600. Raise the systemd timeout to 6h for the longer two-tier run. --- security-review/nightly_sweep.sh | 410 ++++++++++++++++++ security-review/sweep-targets.txt | 15 + .../systemd/sea-haven-secrev.service | 49 +++ .../systemd/sea-haven-secrev.timer | 20 + 4 files changed, 494 insertions(+) create mode 100755 security-review/nightly_sweep.sh create mode 100644 security-review/sweep-targets.txt create mode 100644 security-review/systemd/sea-haven-secrev.service create mode 100644 security-review/systemd/sea-haven-secrev.timer diff --git a/security-review/nightly_sweep.sh b/security-review/nightly_sweep.sh new file mode 100755 index 0000000..999ee3d --- /dev/null +++ b/security-review/nightly_sweep.sh @@ -0,0 +1,410 @@ +#!/usr/bin/env bash +# nightly_sweep.sh — Sea Haven Path B nightly security sweep (R720 / sh-secrev VM). +# +# TWO-TIER, CLEAN-CLONE AUTO-DISCOVERY (no per-repo wiring): +# Discovery: enumerate ALL Sea-Haven-Industries org repos via the GitHub REST API +# (curl + a read-only fine-grained PAT in GH_TOKEN — no gh CLI dependency), then +# mirror each into ~/repo-mirrors as a shallow clean clone (git clone --depth=1, +# default branch from the API). Scanning server-side clones (not developer working +# trees) structurally avoids surfacing local gitignored .env secrets. +# TIER 1 (every repo, every night, $0 Claude): review.sh --scanners-only over every +# mirror — complete deterministic baseline coverage. +# TIER 2 (bounded agentic): the expensive run_headless.py detector+verifier pass runs +# over a deterministic ROUND-ROBIN rotation that fits TOTAL_BUDGET_USD, with a +# persistent cycle pointer so every repo gets a deep pass within MAX_CYCLE_NIGHTS. +# This bounds the draw on the SHARED Max subscription limits (see memory +# reference-claude-subscription-billing): a clean night never scans all repos +# agentically. +# +# Anti-complacency: the canary testbed is ALWAYS scanned agentically first (block + +# recall floor). Reporting is Slack ALARM-ONLY (a clean night posts NOTHING — see +# memory feedback_cloudwatch_alarms). Secret-shaped values are redacted from the Slack +# string; on-disk reports are written mode 600. +# +# Skip a repo: a .security-review-skip file committed at its root, OR an entry in the +# central skip list ($CENTRAL_SKIP_FILE). Repos skipped via their OWN committed marker +# are LOGGED in the report (auditable — a sensitive repo cannot silently self-exclude). +# +# Contract notes: +# - run_headless.py REQUIRES CLAUDE_CODE_OAUTH_TOKEN and pops ANTHROPIC_API_KEY. +# Source ~/secrev.env before invoking (the systemd unit does this via EnvironmentFile). +# - review.sh re-derives the block decision (exit 1 = BLOCK). This script makes NO +# block decision itself; it only reports. +# +# Config (env, all optional except auth): +# GH_TOKEN read-only fine-grained PAT (Contents: read) — REQUIRED for discovery +# GH_ORG org to enumerate (default: Sea-Haven-Industries) +# MIRROR_DIR clean-clone mirror root (default: ~/repo-mirrors) +# CENTRAL_SKIP_FILE one repo name per line, # comments (default: ~/.secrev-skip.txt) +# TARGETS space-separated paths to scan INSTEAD of discovery (manual override) +# TESTBED canary corpus dir (default: ~/security-review-testbed) +# CANARY_FLOOR min confirmed crit+high the canary MUST surface (default: 10) +# TOTAL_BUDGET_USD hard agentic spend ceiling across the night (default: 20) +# PER_TARGET_BUDGET_USD passed to run_headless --total-budget-usd (default: 12) +# MAX_CYCLE_NIGHTS alarm if the agentic rotation hasn't covered every repo in this many nights (default: 4) +# MAX_AGENTIC_PER_NIGHT cap on repos given the deep agentic pass per night, for wall-clock bounding +# (default: 0 = unlimited, bounded only by TOTAL_BUDGET_USD) +# REPORT_ROOT base dir for logs+JSON (default: ~/sweep-reports) +# SLACK_WEBHOOK_URL incoming-webhook URL; if unset, alarms are logged only +# ENABLE_XMODEL_HOOK 1 to run the cross-family critical tiebreak (default: 0) +# ORCHESTRATOR_DIR orchestrator repo root (default: ~/orchestrator) +# VENV_PY python in the SDK venv (default: ~/orchestrator/.venv/bin/python) +# +# Exit: 0 = sweep completed (whether or not it alarmed); 2 = setup/usage error. +set -euo pipefail +export PATH="$HOME/.local/bin:/opt/homebrew/bin:/usr/local/bin:$PATH" + +log() { echo "[nightly_sweep] $*" >&2; } +die() { echo "[nightly_sweep] FATAL: $*" >&2; exit 2; } + +# --- Config + defaults -------------------------------------------------------- +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ORCHESTRATOR_DIR="${ORCHESTRATOR_DIR:-$HOME/orchestrator}" +VENV_PY="${VENV_PY:-$ORCHESTRATOR_DIR/.venv/bin/python}" +RUN_HEADLESS="$HERE/run_headless.py" +REVIEW_SH="$HERE/review.sh" +GH_ORG="${GH_ORG:-Sea-Haven-Industries}" +MIRROR_DIR="${MIRROR_DIR:-$HOME/repo-mirrors}" +CENTRAL_SKIP_FILE="${CENTRAL_SKIP_FILE:-$HOME/.secrev-skip.txt}" +TESTBED="${TESTBED:-$HOME/security-review-testbed}" +CANARY_FLOOR="${CANARY_FLOOR:-10}" +TOTAL_BUDGET_USD="${TOTAL_BUDGET_USD:-20}" +PER_TARGET_BUDGET_USD="${PER_TARGET_BUDGET_USD:-12}" +MAX_CYCLE_NIGHTS="${MAX_CYCLE_NIGHTS:-4}" +MAX_AGENTIC_PER_NIGHT="${MAX_AGENTIC_PER_NIGHT:-0}" +REPORT_ROOT="${REPORT_ROOT:-$HOME/sweep-reports}" +ENABLE_XMODEL_HOOK="${ENABLE_XMODEL_HOOK:-0}" + +command -v jq >/dev/null || die "jq is required" +command -v curl >/dev/null || die "curl is required for org discovery" +command -v git >/dev/null || die "git is required" +[ -x "$VENV_PY" ] || die "venv python not found/executable: $VENV_PY" +[ -f "$RUN_HEADLESS" ] || die "run_headless.py not found: $RUN_HEADLESS" +[ -x "$REVIEW_SH" ] || die "review.sh not found/executable: $REVIEW_SH" +[ -n "${CLAUDE_CODE_OAUTH_TOKEN:-}" ] || die "CLAUDE_CODE_OAUTH_TOKEN not set (source ~/secrev.env)" + +UTC_DATE="$(date -u +%Y-%m-%d)" +UTC_STAMP="$(date -u +%Y-%m-%dT%H:%M:%SZ)" +REPORT_DIR="$REPORT_ROOT/$UTC_DATE" +mkdir -p "$REPORT_DIR"; chmod 700 "$REPORT_ROOT" "$REPORT_DIR" 2>/dev/null || true +ROTATION_STATE="$REPORT_ROOT/.rotation-state.json" +SWEEP_LOG="$REPORT_DIR/sweep.log" +exec > >(tee -a "$SWEEP_LOG") 2>&1 +umask 077 # on-disk reports/logs are not world-readable + +log "=== nightly sweep $UTC_STAMP (two-tier auto-discovery) ===" +log "org=$GH_ORG mirror=$MIRROR_DIR report=$REPORT_DIR total-budget=\$$TOTAL_BUDGET_USD canary-floor=$CANARY_FLOOR" + +# --- Aggregate state ---------------------------------------------------------- +TOTAL_SPEND="0"; BUDGET_HIT=0 +declare -a ALARM_LINES=(); declare -a XMODEL_LINES=(); declare -a MARKER_SKIPS=() +add_spend() { TOTAL_SPEND="$(jq -n --argjson a "$TOTAL_SPEND" --argjson b "${1:-0}" '$a + $b')"; } +over_budget() { jq -n --argjson s "$TOTAL_SPEND" --argjson c "$TOTAL_BUDGET_USD" -e '$c > 0 and $s >= $c' >/dev/null; } + +# --- Secret redaction for the Slack string (defense-in-depth; reports stay on the VM) -- +redact() { + sed -E \ + -e 's/AKIA[0-9A-Z]{16}/AKIA****REDACTED****/g' \ + -e 's/gh[pousr]_[A-Za-z0-9]{20,}/gh*_****REDACTED****/g' \ + -e 's/(xox[baprs]-)[A-Za-z0-9-]{10,}/\1****REDACTED****/g' \ + -e 's/[A-Za-z0-9/+]{40,}/****REDACTED-HIENTROPY****/g' +} + +# --- Discovery: enumerate non-archived org repos via the REST API ------------- +# Emits "nameclone_urldefault_branch" per repo. Returns non-zero on failure. +discover_repos() { + [ -n "${GH_TOKEN:-}" ] || { log "GH_TOKEN unset — cannot enumerate org"; return 1; } + local page=1 got body + while :; do + body="$(curl -fsS \ + -H "Authorization: Bearer $GH_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + -H "X-GitHub-Api-Version: 2022-11-28" \ + "https://api.github.com/orgs/$GH_ORG/repos?per_page=100&type=all&page=$page" 2>>"$REPORT_DIR/discover.log")" || return 1 + echo "$body" | jq -e 'type=="array"' >/dev/null 2>&1 || return 1 + got="$(echo "$body" | jq -r '[.[] | select(.archived==false)] | .[] | [.name, .clone_url, .default_branch] | @tsv')" + [ -n "$got" ] && echo "$got" + [ "$(echo "$body" | jq 'length')" -lt 100 ] && break + page=$((page+1)) + done + return 0 +} + +# --- Mirror one repo as a shallow clean clone ------------------------------------------- +# The token is NEVER persisted to .git/config: the fetch path passes the auth URL inline +# (transient, command-args only), and the clone path scrubs origin immediately after. So a +# failed fetch cannot leave GH_TOKEN at rest on disk. (Residual: the token is briefly visible +# in process args to a local `ps`; acceptable on this single-user unattended box.) +mirror_repo() { # name clone_url default_branch -> 0 ok / 1 fail + local name="$1" url="$2" branch="$3" dir="$MIRROR_DIR/$name" + local auth_url="https://x-access-token:${GH_TOKEN}@${url#https://}" + if [ -d "$dir/.git" ]; then + git -C "$dir" fetch --depth=1 "$auth_url" "$branch" >/dev/null 2>&1 || return 1 + git -C "$dir" reset --hard FETCH_HEAD >/dev/null 2>&1 || return 1 + git -C "$dir" clean -fdq >/dev/null 2>&1 || true + else + git clone --depth=1 --branch "$branch" "$auth_url" "$dir" >/dev/null 2>&1 || return 1 + git -C "$dir" remote set-url origin "$url" >/dev/null 2>&1 || true # clone wrote auth URL → scrub it + fi + return 0 +} + +# --- Skip resolution: "" = scan, else reason ("marker"|"central") -------------- +declare -a CENTRAL_SKIP=() +if [ -f "$CENTRAL_SKIP_FILE" ]; then + while IFS= read -r line; do line="${line%%#*}"; line="$(echo "$line" | xargs || true)" + [ -n "$line" ] && CENTRAL_SKIP+=( "$line" ); done < "$CENTRAL_SKIP_FILE" +fi +skip_reason() { # name dir + local name="$1" dir="$2" + [ -f "$dir/.security-review-skip" ] && { echo "marker"; return; } + for s in ${CENTRAL_SKIP[@]+"${CENTRAL_SKIP[@]}"}; do [ "$s" = "$name" ] && { echo "central"; return; }; done + echo "" +} + +# --- xmodel cross-family critical tiebreak (GUARDED, never fails the sweep) ---- +xmodel_check_criticals() { + local label="$1" result_json="$2" + [ "$ENABLE_XMODEL_HOOK" = "1" ] || return 0 + [ -n "${OPENAI_API_KEY:-}" ] || { log " xmodel hook: OPENAI_API_KEY unset — skipping"; return 0; } + "$VENV_PY" -c 'import langchain_openai' >/dev/null 2>&1 || { log " xmodel hook: deps missing — skipping"; return 0; } + local crits n; crits="$(jq -c '[.findings[]? | select(.status=="confirmed" and .severity=="critical")]' "$result_json" 2>/dev/null || echo '[]')" + n="$(echo "$crits" | jq 'length')"; [ "${n:-0}" -gt 0 ] || return 0 + log " xmodel hook: re-checking $n confirmed critical(s) for $label" + local i=0 + while [ "$i" -lt "$n" ]; do + local summary; summary="$(echo "$crits" | jq -r --argjson i "$i" '.[$i] | "\(.cwe // "n/a") \(.file):\(.line // 0) — \(.title // .id) :: \(.data_flow // "")"')" + local verdict + if verdict="$(cd "$ORCHESTRATOR_DIR" && "$VENV_PY" run.py "Independently assess whether this is a real exploitable vulnerability (yes/no) and why: $summary" 2>>"$REPORT_DIR/xmodel.log")"; then + if echo "$verdict" | grep -qiE '(^|[^a-z])no([^a-z]|$)|not (a |an )?(real |exploitable )?vuln'; then + XMODEL_LINES+=( "DISAGREEMENT on $label critical: $summary (cross_reviewer says NOT a vuln)" ) + fi + else log " xmodel hook: run.py failed for a critical (logged) — continuing"; fi + i=$((i+1)) + done +} + +# --- TIER 1: deterministic scanners over a target dir ------------------------- +# Sets T1_BLOCK/T1_CRIT/T1_HIGH. review.sh exit 0 pass / 1 BLOCK / 2 setup. +T1_BLOCK=0; T1_CRIT=0; T1_HIGH=0 +scan_scanners() { # target slug + local target="$1" slug="$2" result_json="$REPORT_DIR/${slug}.scanners.json" + T1_BLOCK=0; T1_CRIT=0; T1_HIGH=0 + local sup=() + [ -f "$target/.security-review/suppressions.json" ] && sup=(--suppressions "$target/.security-review/suppressions.json") + set +e + "$REVIEW_SH" --scanners-only ${sup[@]+"${sup[@]}"} --json-out "$result_json" "$target" >"$REPORT_DIR/${slug}.scanners.log" 2>&1 + local rc=$? + set -e + [ "$rc" -eq 2 ] && { ALARM_LINES+=( "*$slug*: review.sh scanner setup error. See \`$REPORT_DIR/${slug}.scanners.log\`." ); return; } + T1_CRIT="$(jq -r '(.summary.confirmed_critical // 0)' "$result_json" 2>/dev/null || echo 0)" + T1_HIGH="$(jq -r '(.summary.confirmed_high // 0)' "$result_json" 2>/dev/null || echo 0)" + [ "$rc" -eq 1 ] && T1_BLOCK=1 +} + +# --- TIER 2: agentic run_headless + full review.sh over a target dir ---------- +# Sets LAST_BLOCK/LAST_CRIT/LAST_HIGH/LAST_REASON/LAST_RESULT_JSON/LAST_ERRORS. +LAST_BLOCK=0; LAST_CRIT=0; LAST_HIGH=0; LAST_REASON=""; LAST_RESULT_JSON=""; LAST_ERRORS=0 +scan_agentic() { # target slug + local target="$1" slug="$2" + LAST_BLOCK=0; LAST_CRIT=0; LAST_HIGH=0; LAST_REASON=""; LAST_RESULT_JSON=""; LAST_ERRORS=0 + [ -d "$target" ] || { ALARM_LINES+=( "Target *$slug* ($target) missing — could not scan." ); LAST_ERRORS=1; return; } + local agent_json="$REPORT_DIR/${slug}.agent.json" result_json="$REPORT_DIR/${slug}.result.json" runner_log="$REPORT_DIR/${slug}.runner.log" + LAST_RESULT_JSON="$result_json" + log " [$slug] run_headless.py (per-target budget \$$PER_TARGET_BUDGET_USD)" + if ! "$VENV_PY" "$RUN_HEADLESS" "$target" --out "$agent_json" --total-budget-usd "$PER_TARGET_BUDGET_USD" >>"$runner_log" 2>&1; then + ALARM_LINES+=( "*$slug*: run_headless.py failed (setup error). See \`$runner_log\`." ); LAST_ERRORS=1; return + fi + [ -f "$agent_json" ] || { ALARM_LINES+=( "*$slug*: run_headless produced no JSON." ); LAST_ERRORS=1; return; } + local spend errs; spend="$(jq -r '(._meta.spend_usd // 0)' "$agent_json")"; errs="$(jq -r '(._meta.errors // []) | length' "$agent_json")" + add_spend "$spend"; LAST_ERRORS="$errs" + log " [$slug] spend \$$spend, runner errors $errs, total \$$TOTAL_SPEND" + [ "${errs:-0}" -gt 0 ] && ALARM_LINES+=( "*$slug*: run_headless reported $errs error(s): $(jq -r '(._meta.errors // []) | join("; ")' "$agent_json")" ) + local sup=() + [ -f "$target/.security-review/suppressions.json" ] && sup=(--suppressions "$target/.security-review/suppressions.json") + set +e + "$REVIEW_SH" --agent-findings "$agent_json" ${sup[@]+"${sup[@]}"} --json-out "$result_json" "$target" >"$REPORT_DIR/${slug}.review.log" 2>&1 + local rc=$? + set -e + [ "$rc" -eq 2 ] && { ALARM_LINES+=( "*$slug*: review.sh setup error. See \`$REPORT_DIR/${slug}.review.log\`." ); LAST_ERRORS=$((LAST_ERRORS+1)); return; } + LAST_CRIT="$(jq -r '(.summary.confirmed_critical // 0)' "$result_json" 2>/dev/null || echo 0)" + LAST_HIGH="$(jq -r '(.summary.confirmed_high // 0)' "$result_json" 2>/dev/null || echo 0)" + if [ "$rc" -eq 1 ]; then LAST_BLOCK=1; LAST_REASON="confirmed crit=$LAST_CRIT high=$LAST_HIGH"; log " [$slug] BLOCK ($LAST_REASON)" + else log " [$slug] PASS (crit=$LAST_CRIT high=$LAST_HIGH)"; fi +} + +# ============================== 1) CANARY ==================================== +CANARY_OK=1 +if [ -d "$TESTBED" ]; then + log "--- canary (anti-complacency): $TESTBED ---" + scan_agentic "$TESTBED" "canary" + CANARY_CONFIRMED=0 + if [ -n "$LAST_RESULT_JSON" ] && [ -f "$LAST_RESULT_JSON" ]; then + CANARY_CONFIRMED="$(jq -r '[.findings[]? | select(.status=="confirmed" and (.severity|IN("critical","high")))] | length' "$LAST_RESULT_JSON" 2>/dev/null || echo 0)" + fi + log "canary: block=$LAST_BLOCK confirmed(crit+high)=$CANARY_CONFIRMED (floor=$CANARY_FLOOR)" + if [ "$LAST_BLOCK" -ne 1 ]; then + CANARY_OK=0; ALARM_LINES+=( "*COMPLACENCY ALARM*: canary testbed did NOT block. Result: \`$LAST_RESULT_JSON\`" ) + elif [ "${CANARY_CONFIRMED:-0}" -lt "$CANARY_FLOOR" ]; then + CANARY_OK=0; ALARM_LINES+=( "*COMPLACENCY ALARM*: canary recall $CANARY_CONFIRMED < floor $CANARY_FLOOR. Result: \`$LAST_RESULT_JSON\`" ) + fi + xmodel_check_criticals "canary" "$LAST_RESULT_JSON" +else + CANARY_OK=0; ALARM_LINES+=( "*COMPLACENCY ALARM*: canary testbed missing at $TESTBED." ) +fi + +# ============================== 2) DISCOVER + MIRROR ========================= +declare -a REPO_NAMES=() # scan order (discovery order) +declare -A REPO_DIR=() +if [ -n "${TARGETS:-}" ]; then + # Manual override: scan explicit paths, no discovery/cloning. + # shellcheck disable=SC2206 + arr=( $TARGETS ) + for p in "${arr[@]}"; do + p="${p/#\~/$HOME}"; nm="$(basename "$p")" + REPO_NAMES+=( "$nm" ); REPO_DIR["$nm"]="$p" + done + log "manual TARGETS override: ${REPO_NAMES[*]}" +else + mkdir -p "$MIRROR_DIR" + DISCOVERED="$REPORT_DIR/discovered.tsv" + if discover_repos > "$DISCOVERED" 2>>"$REPORT_DIR/discover.log" && [ -s "$DISCOVERED" ]; then + NREPO="$(wc -l < "$DISCOVERED" | tr -d ' ')" + log "discovered $NREPO non-archived repo(s) in $GH_ORG" + while IFS=$'\t' read -r name url branch; do + [ -n "$name" ] || continue + if mirror_repo "$name" "$url" "$branch"; then + REPO_NAMES+=( "$name" ); REPO_DIR["$name"]="$MIRROR_DIR/$name" + else + log " mirror FAILED: $name"; ALARM_LINES+=( "*$name*: clone/pull failed — not scanned this night. See \`$REPORT_DIR/discover.log\`." ) + fi + done < "$DISCOVERED" + log "mirrored ${#REPO_NAMES[@]} repo(s) into $MIRROR_DIR" + else + ALARM_LINES+=( "*DISCOVERY ALARM*: org enumeration failed (GH_TOKEN missing/invalid or API error). Falling back to existing mirrors; coverage may be stale. See \`$REPORT_DIR/discover.log\`." ) + log "discovery failed — falling back to existing mirrors in $MIRROR_DIR" + if [ -d "$MIRROR_DIR" ]; then + for d in "$MIRROR_DIR"/*/; do [ -d "$d/.git" ] || continue; nm="$(basename "$d")"; REPO_NAMES+=( "$nm" ); REPO_DIR["$nm"]="${d%/}"; done + fi + fi +fi + +# Resolve skips up front (so both tiers honor them and marker-skips are auditable). +declare -a SCANNABLE=() +for nm in ${REPO_NAMES[@]+"${REPO_NAMES[@]}"}; do + reason="$(skip_reason "$nm" "${REPO_DIR[$nm]}")" + if [ "$reason" = "marker" ]; then MARKER_SKIPS+=( "$nm" ); log " skip $nm (repo-committed .security-review-skip)" + elif [ "$reason" = "central" ]; then log " skip $nm (central skip list)" + else SCANNABLE+=( "$nm" ); fi +done +if [ "${#MARKER_SKIPS[@]}" -gt 0 ]; then + ALARM_LINES+=( "*self-excluded repos* (committed .security-review-skip, FYI/audit): ${MARKER_SKIPS[*]}" ) +fi +log "scannable repos: ${#SCANNABLE[@]} (skipped: $(( ${#REPO_NAMES[@]} - ${#SCANNABLE[@]} )))" + +# ============================== 3) TIER 1: scanners over ALL ================== +declare -a BLOCKED_T1=() +for nm in ${SCANNABLE[@]+"${SCANNABLE[@]}"}; do + scan_scanners "${REPO_DIR[$nm]}" "scan-$nm" + if [ "$T1_BLOCK" -eq 1 ]; then + BLOCKED_T1+=( "$nm" ) + ALARM_LINES+=( "*$nm* TIER1/scanners BLOCK: crit=$T1_CRIT high=$T1_HIGH. Result: \`$REPORT_DIR/scan-$nm.scanners.json\`" ) + fi +done +log "tier1 complete: ${#SCANNABLE[@]} scanned, ${#BLOCKED_T1[@]} blocked" + +# ============================== 4) TIER 2: agentic rotation =================== +# Persistent cycle state: {cycle_start, scanned:[names]}. Reset the cycle once every +# scannable repo has had a deep pass; alarm if a cycle runs longer than MAX_CYCLE_NIGHTS. +[ -f "$ROTATION_STATE" ] || echo "{\"cycle_start\":\"$UTC_DATE\",\"scanned\":[]}" > "$ROTATION_STATE" +SCANNED_JSON="$(jq -c '.scanned // []' "$ROTATION_STATE" 2>/dev/null || echo '[]')" +CYCLE_START="$(jq -r '.cycle_start // empty' "$ROTATION_STATE" 2>/dev/null || echo "$UTC_DATE")" +[ -n "$CYCLE_START" ] || CYCLE_START="$UTC_DATE" +if [ "${#SCANNABLE[@]}" -gt 0 ]; then + SCANNABLE_JSON="$(printf '%s\n' "${SCANNABLE[@]}" | jq -R . | jq -cs .)" +else + SCANNABLE_JSON="[]" +fi +# If every scannable repo is already in scanned[], the cycle is complete -> start fresh. +if jq -e -n --argjson sc "$SCANNED_JSON" --argjson all "$SCANNABLE_JSON" '($all - $sc) | length == 0' >/dev/null 2>&1 \ + && [ "$(echo "$SCANNABLE_JSON" | jq 'length')" -gt 0 ]; then + log "agentic rotation: cycle complete ($CYCLE_START) — starting a new cycle" + SCANNED_JSON="[]"; CYCLE_START="$UTC_DATE" +fi +# This night's agentic candidates = scannable repos not yet scanned this cycle, discovery order. +PENDING_JSON="$(jq -c -n --argjson all "$SCANNABLE_JSON" --argjson sc "$SCANNED_JSON" '$all - $sc')" +declare -a BLOCKED_T2=(); AGENTIC_DONE=0 +if over_budget; then + BUDGET_HIT=1; ALARM_LINES+=( "*BUDGET ALARM*: ceiling \$$TOTAL_BUDGET_USD hit after canary (\$$TOTAL_SPEND). No agentic rotation this night." ) +else + while read -r nm; do + [ -n "$nm" ] || continue + if over_budget; then BUDGET_HIT=1; log "budget ceiling hit (\$$TOTAL_SPEND) — pausing rotation"; break; fi + if [ "$MAX_AGENTIC_PER_NIGHT" -gt 0 ] && [ "$AGENTIC_DONE" -ge "$MAX_AGENTIC_PER_NIGHT" ]; then + log "per-night agentic cap ($MAX_AGENTIC_PER_NIGHT) reached — pausing rotation"; break; fi + log "--- agentic: $nm ---" + scan_agentic "${REPO_DIR[$nm]}" "scan-$nm" + SCANNED_JSON="$(echo "$SCANNED_JSON" | jq -c --arg n "$nm" '. + [$n] | unique')" + AGENTIC_DONE=$((AGENTIC_DONE+1)) + if [ "$LAST_BLOCK" -eq 1 ]; then + BLOCKED_T2+=( "$nm" ) + ALARM_LINES+=( "*$nm* TIER2/agentic BLOCK: $LAST_REASON. Result: \`$LAST_RESULT_JSON\`" ) + xmodel_check_criticals "$nm" "$LAST_RESULT_JSON" + fi + done < <(echo "$PENDING_JSON" | jq -r '.[]') +fi +# Persist rotation state. +jq -n --arg cs "$CYCLE_START" --argjson sc "$SCANNED_JSON" '{cycle_start:$cs, scanned:$sc}' > "$ROTATION_STATE" +# Coverage accounting + lag alarm. +REMAINING="$(jq -n --argjson all "$SCANNABLE_JSON" --argjson sc "$SCANNED_JSON" '($all - $sc) | length')" +to_epoch() { date -u -d "$1" +%s 2>/dev/null || date -u -j -f '%Y-%m-%d' "$1" +%s 2>/dev/null || echo 0; } +CYCLE_AGE=$(( ( $(to_epoch "$UTC_DATE") - $(to_epoch "$CYCLE_START") ) / 86400 )) +log "agentic rotation: scanned $AGENTIC_DONE this night, $REMAINING still pending in cycle (started $CYCLE_START, age ${CYCLE_AGE}d)" +if [ "$REMAINING" -gt 0 ] && [ "$CYCLE_AGE" -ge "$MAX_CYCLE_NIGHTS" ]; then + ALARM_LINES+=( "*COVERAGE ALARM*: agentic rotation behind — $REMAINING repo(s) not deep-scanned in ${CYCLE_AGE}d (cycle since $CYCLE_START, max $MAX_CYCLE_NIGHTS). Raise budget or check for failures." ) +fi + +# Fold xmodel disagreements into the alarm set. +for x in ${XMODEL_LINES[@]+"${XMODEL_LINES[@]}"}; do ALARM_LINES+=( "$x" ); done + +# ============================== 5) ALARM-ONLY REPORT ========================= +ALARM=0 +[ "${#BLOCKED_T1[@]}" -gt 0 ] && ALARM=1 +[ "${#BLOCKED_T2[@]}" -gt 0 ] && ALARM=1 +[ "$CANARY_OK" -ne 1 ] && ALARM=1 +[ "$BUDGET_HIT" -eq 1 ] && ALARM=1 +[ "${#ALARM_LINES[@]}" -gt 0 ] && ALARM=1 + +SUMMARY_LINE="sweep $UTC_STAMP: scannable=${#SCANNABLE[@]} tier1_blocked=${#BLOCKED_T1[@]} tier2_scanned=$AGENTIC_DONE tier2_blocked=${#BLOCKED_T2[@]} canary_ok=$CANARY_OK spend=\$$TOTAL_SPEND/\$$TOTAL_BUDGET_USD alarm=$ALARM report=$REPORT_DIR" +echo "$SUMMARY_LINE" + +if [ "$ALARM" -ne 1 ]; then + log "clean night — no alarm conditions. Posting NOTHING to Slack (ALARM-only policy)." + exit 0 +fi + +ALARM_BODY="$(printf '%s\n' ${ALARM_LINES[@]+"${ALARM_LINES[@]}"} | sed 's/^/• /')" +SLACK_TEXT=":rotating_light: *Sea Haven nightly security sweep — ALARM* ($UTC_STAMP) +$ALARM_BODY + +Coverage: tier1 scanners ${#SCANNABLE[@]} repos · tier2 agentic $AGENTIC_DONE this night ($REMAINING pending) · canary_ok=$CANARY_OK +Spend: \$$TOTAL_SPEND (ceiling \$$TOTAL_BUDGET_USD) +Reports + JSON: \`$REPORT_DIR\` (on sh-secrev VM)" +SLACK_TEXT="$(echo "$SLACK_TEXT" | redact)" + +log "ALARM conditions present — composing Slack post" +echo "$SLACK_TEXT" >&2 + +if [ -n "${SLACK_WEBHOOK_URL:-}" ] && command -v curl >/dev/null; then + PAYLOAD="$(jq -n --arg t "$SLACK_TEXT" '{text:$t}')" + if curl -fsS -X POST -H 'Content-Type: application/json' --data "$PAYLOAD" "$SLACK_WEBHOOK_URL" >/dev/null 2>>"$REPORT_DIR/slack.log"; then + log "Slack alarm posted." + else + log "Slack POST FAILED — see $REPORT_DIR/slack.log. Alarm text is in $SWEEP_LOG." + fi +else + log "SLACK_WEBHOOK_URL unset (or curl missing) — alarm logged to $SWEEP_LOG only." +fi + +# An alarm is a reportable condition, not a script crash. Exit 0 so systemd shows success. +exit 0 diff --git a/security-review/sweep-targets.txt b/security-review/sweep-targets.txt new file mode 100644 index 0000000..9be820e --- /dev/null +++ b/security-review/sweep-targets.txt @@ -0,0 +1,15 @@ +# sweep-targets.txt — one repo path per line for the nightly Path B sweep. +# Lines starting with '#' and blank lines are ignored. ~ is expanded. +# Override at runtime with the TARGETS env var (space-separated paths). +# +# NOTE: the testbed canary corpus is ALWAYS scanned by nightly_sweep.sh as the +# anti-complacency check; do NOT list it here (it is handled separately). +# +# TODO (Phase 5): add the first real hardened repo here once it is cloned on the +# VM (candidates: payments-dashboard / proposal-system / procurement-ingest). +# +# Deliberately EMPTY by default = canary-only nights. Do NOT scan ~/orchestrator: +# it holds ~/orchestrator/.env with live provider API keys, which the agentic +# detector could surface into sweep reports / Slack. Only add repos with no +# plaintext secrets (or scrub/exclude secret files first). +# ~/orchestrator diff --git a/security-review/systemd/sea-haven-secrev.service b/security-review/systemd/sea-haven-secrev.service new file mode 100644 index 0000000..3e40fb0 --- /dev/null +++ b/security-review/systemd/sea-haven-secrev.service @@ -0,0 +1,49 @@ +# sea-haven-secrev.service — Path B nightly security sweep (sh-secrev VM, user adam). +# +# Install (on the VM, as root): +# sudo cp sea-haven-secrev.service /etc/systemd/system/ +# sudo cp sea-haven-secrev.timer /etc/systemd/system/ +# sudo systemctl daemon-reload +# sudo systemctl enable --now sea-haven-secrev.timer # timer drives the run; do NOT enable the .service +# systemctl list-timers sea-haven-secrev.timer # confirm next run +# sudo systemctl start sea-haven-secrev.service # optional: run once now to smoke-test +# journalctl -u sea-haven-secrev.service -e # logs (also under ~/sweep-reports//) +# +# Secrets come from the EnvironmentFiles (the leading '-' = optional, no failure if absent): +# ~/secrev.env -> CLAUDE_CODE_OAUTH_TOKEN (required by run_headless.py), +# GH_TOKEN (read-only fine-grained PAT — REQUIRED for org auto-discovery), +# SLACK_WEBHOOK_URL +# ~/orchestrator/.env -> OPENAI_API_KEY etc. (only needed if ENABLE_XMODEL_HOOK=1) +# +# GH_TOKEN must be a fine-grained PAT scoped to the Sea-Haven-Industries org with READ-ONLY +# Contents (and Metadata) permission — nothing else. It enumerates repos and clones them into +# ~/repo-mirrors. Never give this unattended box a write-capable token. + +[Unit] +Description=Sea Haven Path B nightly security sweep +After=network-online.target +Wants=network-online.target + +[Service] +Type=oneshot +User=adam +WorkingDirectory=/home/adam/orchestrator +EnvironmentFile=-/home/adam/secrev.env +EnvironmentFile=-/home/adam/orchestrator/.env +# Tune ceilings/targets here without editing the script (uncomment to override defaults): +# Environment=TOTAL_BUDGET_USD=20 +# Environment=PER_TARGET_BUDGET_USD=12 +# Environment=CANARY_FLOOR=10 +# Environment=MAX_CYCLE_NIGHTS=4 +# Environment=MAX_AGENTIC_PER_NIGHT=0 +# Environment=GH_ORG=Sea-Haven-Industries +# Environment=MIRROR_DIR=/home/adam/repo-mirrors +# Environment=ENABLE_XMODEL_HOOK=0 +ExecStart=/home/adam/orchestrator/security-review/nightly_sweep.sh +# Two-tier sweep (scanners over every repo + a budget-bounded agentic rotation) runs for hours; +# 6h ceiling bounds a hang without killing a healthy long night. Spend is capped by TOTAL_BUDGET_USD. +TimeoutStartSec=21600 +Nice=10 + +[Install] +WantedBy=multi-user.target diff --git a/security-review/systemd/sea-haven-secrev.timer b/security-review/systemd/sea-haven-secrev.timer new file mode 100644 index 0000000..b377585 --- /dev/null +++ b/security-review/systemd/sea-haven-secrev.timer @@ -0,0 +1,20 @@ +# sea-haven-secrev.timer — fires the nightly sweep at ~02:00 local, user adam. +# +# Install: see the header of sea-haven-secrev.service. In short: +# sudo systemctl enable --now sea-haven-secrev.timer +# systemctl list-timers sea-haven-secrev.timer +# +# Persistent=true → if the VM was off at 02:00, the sweep runs at next boot. +# RandomizedDelaySec spreads load off an exact-minute spike. + +[Unit] +Description=Run the Sea Haven Path B security sweep nightly (~02:00) + +[Timer] +OnCalendar=*-*-* 02:00:00 +Persistent=true +RandomizedDelaySec=600 +Unit=sea-haven-secrev.service + +[Install] +WantedBy=timers.target -- 2.50.1 From af5ce7362e97838c8dcfbd4710ef3d5b9eec36b2 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:00:00 -0400 Subject: [PATCH 04/16] Suppress orchestrator's own .env.example gitleaks false positive gitleaks git-mode scans committed history, so the .env.example placeholder flags from history and would block the orchestrator's own pre-push gate. Suppress it with a written justification. Also gitignore stray .adf_final*.json left by an unrelated tool. --- .gitignore | 3 +++ .security-review/suppressions.json | 8 ++++++++ 2 files changed, 11 insertions(+) create mode 100644 .security-review/suppressions.json diff --git a/.gitignore b/.gitignore index ce218a5..428cb36 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,6 @@ __pycache__/ .pytest_cache/ .ruff_cache/ .DS_Store + +# Stray Atlassian Document Format exports left by an unrelated tool — not part of this repo. +.adf_final*.json diff --git a/.security-review/suppressions.json b/.security-review/suppressions.json new file mode 100644 index 0000000..56cee62 --- /dev/null +++ b/.security-review/suppressions.json @@ -0,0 +1,8 @@ +{ + "suppressions": [ + { + "id": "gitleaks-generic-api-key-2", + "justification": "False positive. .env.example:2 is a documented placeholder token (not a live secret) that exists to show the required env var shape. gitleaks runs in git-mode and scans committed history, so it flags the placeholder even though the working-tree value is inert. No real credential is or was exposed. Reviewed 2026-06-16." + } + ] +} -- 2.50.1 From a05afb5ebc9b254dcff4162cc46f0be5461af344 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:00:00 -0400 Subject: [PATCH 05/16] Update security-review docs for two-tier auto-discovery Rewrite README + DEPLOY-R720 for the clean-clone mirror model, the read-only GH_TOKEN PAT recipe, the global hook installer, and the no-CI-by-design decision. --- security-review/DEPLOY-R720.md | 184 ++++++++++++++++++++++++--------- security-review/README.md | 114 ++++++++++++++++---- 2 files changed, 232 insertions(+), 66 deletions(-) diff --git a/security-review/DEPLOY-R720.md b/security-review/DEPLOY-R720.md index 24fde9e..4f1259b 100644 --- a/security-review/DEPLOY-R720.md +++ b/security-review/DEPLOY-R720.md @@ -1,54 +1,144 @@ -# Phase 3 — Path B deployment (R720 host + CI backstop) +# Phase 3 — Path B deployment (R720 VM) -Status: **design + scripts; NOT deployed.** Needs the R720 (hostname SH-NVR, deprecated NVR, no GPU) -and Adam's hands-on involvement. Nothing here has been run against the box. See memory +Status: **host built; headless runner built + validated; two-tier auto-discovery nightly sweep built.** +Pending: provision the read-only `GH_TOKEN` and run one live VM dry-run to validate the clone-mirror path +end-to-end. **CI was removed by design** — the git hooks + this nightly sweep are the backstop. See memory `project-security-review-agent`. -## What Phase 3 delivers -1. The orchestrator hosted on the R720 as an always-on service (the Path B execution engine). -2. A **headless agentic runner** (`run_headless.py`) so the detector fan-out + proof-or-kill verifier - can run unattended (interactive `/sh-security-review` can't run in CI/cron). -3. CI backstop (`ci/security-review.yml`) running the same `review.sh` on every PR. -4. Nightly full-repo sweep + planted canary vulns + two-model disagreement on criticals. -5. Budget/telemetry ceiling against the Max-20x **$200/mo Agent SDK credit pool** - (see memory `reference-claude-subscription-billing`). +Path B is the unattended backstop that shares one pure-code gate (`review.sh`) with the interactive Path A +(`/sh-security-review`). This file is the operator runbook for the box that runs it. -## The load-bearing piece to BUILD first: `run_headless.py` -The interactive skill orchestrates subagents via the Claude Code Agent tool (Max-covered). Headless, -that must become explicit API/SDK calls. `run_headless.py` should: -- take `--scope` + target dir, read the in-scope files; -- run the 6 detectors (injection/authz/secrets-crypto/iac-iam/web-client/logic) as parallel calls, - each fresh context, using the SAME prompts as `~/.claude/commands/sh-security-review.md`; -- run the proof-or-kill verifier pass; -- emit the finding schema JSON that `review.sh --agent-findings` consumes. -- **Billing:** authenticate via the **Claude Agent SDK with the subscription** to spend the $200/mo - pool before API overflow (NOT a raw key) — see the billing memory. Non-Claude cross-family tiebreak - stays on Bedrock/API. ⚠️ This code cannot be validated until it runs with real creds; do not mark - done until tested. +## Host -## R720 host setup (Windows Server 2022, no GPU) -Recommended: run the Python orchestrator in **WSL2 or Docker** rather than native Windows Python. -1. `git clone https://github.com/amoussa1229/orchestrator` onto the box. -2. Python 3.12 + `pip install -r requirements.txt` (+ the Agent SDK). -3. Install the deterministic scanners (semgrep/gitleaks/checkov/cfn-lint/pip-audit) on the box. -4. Secrets on the box (NOT in git): Agent SDK subscription auth, Bedrock creds, Slack token, repo PATs. -5. Run as a service (NSSM on Windows, or systemd inside WSL2). Expose to the harness as an - MCP/HTTP service (mirror the existing unifi MCP pattern) OR keep it as a local scheduled job. -6. Scheduler: nightly full-repo sweep → `review.sh` → Slack report (ALARM-style, see - memory `feedback_cloudwatch_alarms`: only notify on findings, not clean runs). +- **Hypervisor:** R720 at `10.10.60.40` (Windows Server 2022, Hyper-V role). +- **VM:** `sh-secrev`, always-on Ubuntu 24.04 (kernel 6.8), Gen2, 4GB / 2 vCPU / 40GB dynamic vhdx. +- **Reach it:** `ssh -i ~/.ssh/r720_seahaven adam@10.10.60.120` (key-only, NOPASSWD sudo). -## Anti-complacency reinforcements (Phase 3) -- **Canary vulns:** keep the testbed's planted vulns in a fixture the nightly sweep also scans; if the - agent ever misses a known canary, that's a complacency signal → alert. -- **Two-model disagreement:** on confirmed criticals, run the cross-family reviewer (orchestrator - GPT-4.1) and flag disagreement for human review. -- **Budget guard:** track output tokens vs the $200/mo pool; stop/alert before overflow. +Operate on the VM, not from the Mac against the host by hand. -## Open dependencies (Adam) -- R720 reachable + WSL2/Docker chosen. -- GitHub repo/org secrets: `ANTHROPIC_API_KEY` (or SDK auth), set `vars.SECURITY_REVIEW_AGENTIC=1` - once `run_headless.py` is tested. -- Decide: promote `ci/security-review.yml` into Sea-Haven-Industries/.github as a reusable workflow - (per engineering-handbook/cicd.md) vs per-repo copy. -- checkov severity filtering (it emitted 51-55 best-practice mediums on payments-dashboard — tune - before nightly or the Slack reports are noise). +## What is installed on the VM + +- **Deterministic scanners:** semgrep, gitleaks, checkov, pip-audit, cfn-lint. **Node 18** (`npm audit`). +- **`claude` CLI** (Node) — the subscription-auth path for Path B. +- **Python 3.12 venv** at `~/orchestrator/.venv` with `claude-agent-sdk`. +- **Repo:** `~/orchestrator/` (rsync from the Mac, `.env` excluded — NOT a git clone). After editing the + sweep locally, re-sync: `rsync -av --exclude .env --exclude .venv ~/Documents/repositories/orchestrator/ adam@10.10.60.120:orchestrator/`. +- **Testbed corpus:** `~/security-review-testbed` (also rsync'd; includes the Node + .NET fixtures). +- **No `gh` CLI required** — discovery uses the GitHub REST API via `curl`. `run_headless.py` is + self-contained (detector/verifier prompts are inline), so the VM needs no `~/.claude` assets to run. + +## Auth, billing, and the read-only GitHub token + +### Claude (subscription OAuth) +- Token from `claude setup-token`, stored in `~/secrev.env` as `CLAUDE_CODE_OAUTH_TOKEN` (mode 600, NOT in git). +- The 2026-06-15 SDK-billing split was **deferred**, so automated SDK usage draws from the Max 20x + subscription's normal usage limits — the same pool as interactive Claude Code. The two-tier sweep below + is what keeps that draw bounded. See memory `reference-claude-subscription-billing`. +- **CRITICAL:** a raw `ANTHROPIC_API_KEY` would silently win and meter to API rates — it must NOT be set on + this host. `run_headless.py` pops it defensively and refuses to run without `CLAUDE_CODE_OAUTH_TOKEN`. + +### GitHub (`GH_TOKEN`, read-only — REQUIRED for auto-discovery) +The nightly sweep enumerates and clones org repos with a **fine-grained, read-only PAT**. Never give this +always-on box a write-capable token. + +1. github.com → Settings → Developer settings → **Fine-grained personal access tokens** → Generate new. +2. **Resource owner:** Sea-Haven-Industries. **Repository access:** All repositories. +3. **Permissions:** Repository → **Contents: Read-only**, **Metadata: Read-only** (auto). Nothing else. +4. Set an expiry (e.g. 90 days; calendar a rotation). Generate and copy the `github_pat_...` value. +5. On the VM, append it to `~/secrev.env` and lock the file down: + ``` + echo 'GH_TOKEN=github_pat_xxxxxxxx' >> ~/secrev.env && chmod 600 ~/secrev.env + ``` +6. Verify (should print repo names, not a 401): + ``` + set -a; . ~/secrev.env; set +a + curl -fsS -H "Authorization: Bearer $GH_TOKEN" \ + "https://api.github.com/orgs/Sea-Haven-Industries/repos?per_page=3" | jq '.[].full_name' + ``` + +### Non-Claude provider keys +The GPT-4.1 critical tiebreak (optional) uses keys in `~/orchestrator/.env` (mode 600, gitignored, +auto-loaded by `run.py`). They bill to their own provider accounts — keep them out of `~/secrev.env`. + +## The headless runner: `run_headless.py` + +Runs the 6 fresh-context detectors + proof-or-kill verifier unattended via the Agent SDK; emits the +finding-schema JSON that `review.sh --agent-findings` consumes. Read-only tools, hermetic +(`setting_sources=[]`), fails toward over-reporting. CLI: + +``` +CLAUDE_CODE_OAUTH_TOKEN=... python3 run_headless.py TARGET_DIR \ + [--scope "src infra web"] [--out findings.json] [--model claude-...] \ + [--detectors injection,authz,...] [--concurrency 3] [--max-turns 40] \ + [--detector-budget-usd 2.0] [--total-budget-usd 12.0] +``` +When the total budget is exhausted the verifier is skipped and remaining candidates stay `unverified` — +never silently dropped. Manual single-repo run: +``` +cd ~/orchestrator +set -a; . ~/secrev.env; set +a +.venv/bin/python security-review/run_headless.py ~/security-review-testbed --out /tmp/agent.json +security-review/review.sh --agent-findings /tmp/agent.json ~/security-review-testbed +``` + +## Nightly two-tier, clean-clone auto-discovery sweep + +`nightly_sweep.sh` needs **no per-repo wiring**. Each night it: + +1. **Discovers** every non-archived Sea-Haven-Industries repo via the REST API (`curl` + `GH_TOKEN`) and + **mirrors** each as a shallow clean clone (`git clone --depth=1`, default branch from the API + `default_branch`) into `~/repo-mirrors`. The token is injected only for the fetch and scrubbed from the + on-disk remote afterward. Clean clones contain no developer-local gitignored `.env`, so live secrets + stay out of scope by construction. +2. **Canary first:** scans `~/security-review-testbed` agentically (anti-complacency) — must block and meet + the recall floor, else COMPLACENCY ALARM. +3. **Tier 1 (every repo, $0 Claude):** `review.sh --scanners-only` over every mirror. +4. **Tier 2 (bounded agentic):** `run_headless.py` over a deterministic round-robin rotation that fits + `TOTAL_BUDGET_USD`, with a persistent cycle pointer (`~/sweep-reports/.rotation-state.json`) so every + repo gets a deep pass within `MAX_CYCLE_NIGHTS`; a COVERAGE ALARM fires if it falls behind. + +ALARM-only (a clean night posts nothing). Secret-shaped values are redacted from the Slack string; reports +under `~/sweep-reports//` are mode 600. + +### Config (env / systemd `Environment=`) +`GH_ORG` (Sea-Haven-Industries) · `MIRROR_DIR` (~/repo-mirrors) · `TOTAL_BUDGET_USD` (20) · +`PER_TARGET_BUDGET_USD` (12) · `CANARY_FLOOR` (10) · `MAX_CYCLE_NIGHTS` (4) · `MAX_AGENTIC_PER_NIGHT` +(0 = unlimited) · `CENTRAL_SKIP_FILE` (~/.secrev-skip.txt) · `ENABLE_XMODEL_HOOK` (0) · +`TARGETS` (manual override — scan explicit paths, no discovery). + +### Skip a repo +Commit a `.security-review-skip` at its root, **or** add its name to `~/.secrev-skip.txt`. Marker-skips are +logged in the report (a sensitive repo cannot silently self-exclude). + +### Manual dry-run (do this once after provisioning `GH_TOKEN`) +``` +cd ~/orchestrator +set -a; . ~/secrev.env; set +a +./security-review/nightly_sweep.sh +# Watch: discovery count, mirrors, canary block+recall, tier1 over all repos, tier2 rotation, clean exit. +# Then re-tune CANARY_FLOOR to the reported recall, and confirm the ALARM path with a forced failure. +``` + +### Install the timer +``` +sudo cp security-review/systemd/sea-haven-secrev.{service,timer} /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now sea-haven-secrev.timer # the timer drives it; do not enable the .service +systemctl list-timers sea-haven-secrev.timer +``` +Fires nightly ~02:00 local (`Persistent=true` catches missed runs). `TimeoutStartSec=21600` (6h) bounds a +hang without killing a healthy long night; spend is capped by `TOTAL_BUDGET_USD`. + +## Anti-complacency reinforcements +- **Canary:** the testbed (now Python/IaC/React + Node + .NET planted vulns) is scanned every night; a + recall drop or non-block is a COMPLACENCY ALARM. +- **Coverage:** the rotation pointer + `MAX_CYCLE_NIGHTS` guarantee every repo gets a deep pass on a cadence, + with a COVERAGE ALARM if it slips — no silent incomplete coverage. +- **Two-model disagreement (optional):** `ENABLE_XMODEL_HOOK=1` re-checks confirmed criticals with GPT-4.1. + +## Remaining (deferred by design) +- **Persistent budget/telemetry ledger:** cross-run spend tracking beyond the per-run + nightly caps (optional). +- **Phase 4 roster growth** (compliance/drift sweep, CVE agent, optional auto-fixer) — only per a real job. +- **Phase 5 remediation:** harden findings as real repos surface them (payments-dashboard first). +- **Confluence:** document `sh-secrev` as standing infrastructure (always-on VM holding a read-only org PAT, + pulling all org repos nightly) in the IT host/LAN inventory. diff --git a/security-review/README.md b/security-review/README.md index b542167..dc23536 100644 --- a/security-review/README.md +++ b/security-review/README.md @@ -1,32 +1,108 @@ # security-review -The Sea Haven security-review gate. One pure-code script (`review.sh`), many triggers. +The Sea Haven security-review gate. One pure-code script (`review.sh`) is the decision-maker; everything +else (hooks, the interactive skill, the headless runner, the nightly sweep) is a trigger that feeds it. See memory `project-security-review-agent` for the full design. ## Pieces -- `review.sh` — merges deterministic-scanner findings + agent findings, dedups, applies - suppressions (justification required), and makes the **block decision** (no agent decides). Exit 1 = BLOCK. -- `hooks/pre-commit` + `install-hooks.sh` — fast scanners-only hook for a target repo. -- The agentic detector/verifier pass is the interactive `/sh-security-review` slash command - (`~/.claude/commands/sh-security-review.md`), schema at `~/.claude/security-review/finding.schema.json`. +- `review.sh` — merges deterministic-scanner findings + agent findings, dedups, applies suppressions + (justification required), and makes the **block decision** (no agent decides). Exit 1 = BLOCK. +- `hooks/pre-commit`, `hooks/pre-push` + `install-hooks.sh` — fast `--scanners-only` gates. Install once + globally for every repo, or per-repo (see below). +- `skill/sh-security-review.md` — the interactive agentic detector/verifier prompt (Path A, Max-covered). + `finding.schema.json` — the structured finding contract both paths emit. `install-hooks.sh --global` + links these into `~/.claude/` (this repo is the source of truth). +- `run_headless.py` — the Path B headless detector fan-out + proof-or-kill verifier (Claude Agent SDK, + subscription OAuth). Self-contained: prompts are inline, so the VM needs no `~/.claude` assets to run it. +- `nightly_sweep.sh` + `systemd/` — the unattended two-tier sweep on the `sh-secrev` VM (R720). ## Triggers (one script, many entry points) - **On-demand (primary):** run `/sh-security-review` in a Claude Code session (Max-covered), have it - write its schema JSON, then `review.sh --agent-findings out.json ` to gate. -- **Pre-commit:** `install-hooks.sh ` — fast deterministic scanners abort the commit early. -- **CI (Phase 3):** the same `review.sh` runs headless as the unbypassable backstop. + write its schema JSON, then `review.sh --agent-findings out.json ` to gate. Required before + pushing payments/auth/IaC/input-handling changes (see the global CLAUDE.md security-review rule). +- **Pre-commit / pre-push:** the global git hooks run deterministic scanners automatically. +- **Nightly:** the VM sweep (Path B) is the unattended backstop. + +### Installing the hooks +``` +# Global — gate EVERY repo on this machine, and link the skill + schema into ~/.claude: +security-review/install-hooks.sh --global + +# Per-repo — for a repo that sets its own core.hooksPath (e.g. husky) and would shadow the global hook: +security-review/install-hooks.sh /path/to/repo +``` +The global mode sets `git config --global core.hooksPath ~/.config/git/hooks`. Skip a repo with a +`.security-review-skip` file at its root; suppress a specific false positive in the repo's +`.security-review/suppressions.json` (a written justification is required and is surfaced); bypass once +with `git push --no-verify`. Caveat: a repo with its own local `core.hooksPath` overrides the global hook +— install per-repo there. See memory `reference_global_security_review_hook`. + +## No CI — by design +There is **no CI** wiring for this gate. For a solo dev the git hooks + nightly VM sweep are the backstop, +so the parked CI drafts (`ci/*.yml`, `CI-BACKSTOP-NOTES.md`) were removed; recover them from git history +(the commit that deleted `security-review/ci/`) if the team ever goes multi-dev. The orchestrator repo's +own `ci.yaml` (ruff + tests) is unrelated and stays. ## Scanners -`review.sh` runs whatever is installed and logs the rest with install commands (no silent skips). -Currently wired: `cfn-lint`. To give the deterministic layer teeth, install: +`review.sh` runs whatever is installed and logs the rest with install commands (no silent skips): +`semgrep` (`p/security-audit` + `p/secrets` + `p/javascript`), `gitleaks` (git-mode — scans committed +history, respects `.gitignore`), `checkov`, `cfn-lint`, `pip-audit`, `npm audit`. Each is normalized into +the finding schema. Install the full set: ``` -pipx install semgrep # SAST: injection / authz / xss -brew install gitleaks # hardcoded secrets -pipx install pip-audit # vulnerable Python deps -pipx install checkov # IaC / IAM misconfig +pipx install semgrep pip-audit checkov # SAST / vulnerable Python deps / IaC misconfig +brew install gitleaks # hardcoded secrets +# cfn-lint via pip; Node.js provides npm audit ``` -Each needs a small normalizer added to `review.sh` (map its JSON to the finding schema) when installed. -## Status -review.sh gate validated against the local testbed: 16 findings, correct dedup of distinct -same-CWE findings, suppression-without-justification rejected and surfaced, BLOCK on confirmed crit/high. +## Nightly sweep (Path B) — two-tier, clean-clone auto-discovery +`nightly_sweep.sh` runs on the `sh-secrev` Ubuntu VM (R720) and needs **no per-repo wiring**. It: + +1. **Discovers** every non-archived Sea-Haven-Industries repo via the GitHub REST API (`curl` + a + read-only `GH_TOKEN`; no `gh` CLI dependency) and **mirrors** each as a shallow clean clone + (`git clone --depth=1`, default branch from the API) into `~/repo-mirrors`. Scanning server-side + clones — not developer working trees — structurally keeps local gitignored `.env` secrets out of scope. +2. **Tier 1 (every repo, every night, $0 Claude):** `review.sh --scanners-only` over every mirror — + complete deterministic baseline coverage. +3. **Tier 2 (bounded agentic):** `run_headless.py` over a deterministic **round-robin rotation** that + fits `TOTAL_BUDGET_USD`, with a persistent cycle pointer so every repo gets a deep pass within + `MAX_CYCLE_NIGHTS`. This bounds the draw on the shared Max limits (a clean night never deep-scans all + repos). A `COVERAGE ALARM` fires if the rotation falls behind. + +It is **ALARM-only**: a clean night posts nothing. See memory `feedback_cloudwatch_alarms`. + +### Skip / override +- A repo is skipped if it commits a `.security-review-skip` marker **or** is listed in the central skip + file (`~/.secrev-skip.txt`, one repo name per line). Repos skipped via their own committed marker are + **logged in the report** so a sensitive repo can't silently self-exclude. +- `TARGETS="/path/a /path/b"` overrides discovery entirely (scan explicit paths, no cloning). +- The canary corpus (`~/security-review-testbed`) is **always** scanned agentically first as the + anti-complacency check — independent of the skip filter. + +### Anti-complacency + guards +- **Canary check:** the testbed MUST block AND surface ≥ `CANARY_FLOOR` (default 10) confirmed crit/high. + Otherwise → COMPLACENCY ALARM. The corpus now includes Node + .NET fixtures (see the testbed key); + re-tune the floor after the first VM canary run reports the expanded recall number. +- **Budget ceiling:** `TOTAL_BUDGET_USD` (default 20) caps aggregate agentic spend; `PER_TARGET_BUDGET_USD` + (default 12) caps each repo; `MAX_AGENTIC_PER_NIGHT` (default 0 = unlimited) optionally caps wall-clock. + With the SDK-billing split deferred (memory `reference-claude-subscription-billing`), spend draws from + the Max subscription limits, so the two-tier design keeps full coverage cheap and bounds the agentic draw. +- **Two-model hook (optional, off):** `ENABLE_XMODEL_HOOK=1` re-checks each confirmed CRITICAL with the + orchestrator cross-family reviewer (GPT-4.1) and flags disagreement; skips gracefully, never fails the sweep. +- **Redaction:** secret-shaped values are masked in the Slack ALARM string; on-disk reports are mode 600. + +### Secrets / env (`~/secrev.env`, mode 600) +- `CLAUDE_CODE_OAUTH_TOKEN` — required (`run_headless.py` pops `ANTHROPIC_API_KEY`). +- `GH_TOKEN` — **read-only fine-grained PAT** scoped to the org (Contents + Metadata: read-only, nothing + else) for discovery + cloning. Never give this unattended box a write-capable token. +- `SLACK_WEBHOOK_URL` — alarms (plain incoming-webhook). `~/orchestrator/.env` → `OPENAI_API_KEY` (xmodel hook only). +- Reports + per-target JSON land under `~/sweep-reports//`. + +### Install the timer +Units are in `systemd/`; full runbook is `DEPLOY-R720.md`. On the VM: +``` +sudo cp systemd/sea-haven-secrev.{service,timer} /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now sea-haven-secrev.timer # the timer drives it; do not enable the .service +sudo systemctl start sea-haven-secrev.service # optional one-off smoke test +``` +The timer fires nightly at ~02:00 local (`Persistent=true` catches missed runs after downtime). -- 2.50.1 From f4dd72ced9643ba1d147b0a5a6565f4e6727ba5c Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:00:10 -0400 Subject: [PATCH 06/16] Add headless detector fan-out + proof-or-kill verifier runner run_headless.py is the Path B engine the nightly sweep calls: the 6 fresh-context detectors + proof-or-kill verifier from /sh-security-review, run unattended via the Claude Agent SDK on subscription OAuth (pops ANTHROPIC_API_KEY so the API key can't silently win). Read-only tools, hermetic, per-call + total budget caps, fails toward over-reporting. Emits the finding schema that review.sh --agent-findings consumes. --- security-review/run_headless.py | 447 ++++++++++++++++++++++++++++++++ 1 file changed, 447 insertions(+) create mode 100644 security-review/run_headless.py diff --git a/security-review/run_headless.py b/security-review/run_headless.py new file mode 100644 index 0000000..178e249 --- /dev/null +++ b/security-review/run_headless.py @@ -0,0 +1,447 @@ +#!/usr/bin/env python3 +"""run_headless.py — Path B headless detector fan-out + proof-or-kill verifier. + +Reuses the /sh-security-review detector + verifier prompts, but runs them +unattended via the Claude Agent SDK instead of interactive Claude Code subagents. +Authenticates with the Claude subscription OAuth token (CLAUDE_CODE_OAUTH_TOKEN) +through the bundled `claude` CLI — NEVER a raw ANTHROPIC_API_KEY (which would be +metered and would silently win if both were set, so we pop it). + +Emits the finding-schema JSON ({findings, summary}) that +`review.sh --agent-findings` consumes. review.sh re-derives the gate, so this +script's job is high-recall candidate generation + proof-or-kill verification, +failing toward over-reporting (never silently drops a candidate or a parse error). + +See memory project-security-review-agent and security-review/DEPLOY-R720.md. + +Usage: + CLAUDE_CODE_OAUTH_TOKEN=... python3 run_headless.py TARGET_DIR \ + [--scope "src infra web"] [--out findings.json] [--model claude-...] \ + [--concurrency 3] [--detectors injection,authz] \ + [--detector-budget-usd 2.0] [--total-budget-usd 12.0] +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import os +import re +import sys +import time +from pathlib import Path + +from claude_agent_sdk import ClaudeAgentOptions, query + +# Read-only surface: detectors reason over source, they don't mutate or fetch. +READONLY_TOOLS = ["Read", "Grep", "Glob"] +BLOCKED_TOOLS = ["Bash", "Write", "Edit", "NotebookEdit", "WebFetch", "WebSearch"] + +# Detector category -> closed checklist (verbatim intent from sh-security-review.md). +DETECTORS: dict[str, str] = { + "injection": "SQL/command/template injection, unsafe deserialization, SSRF, path/file traversal, XXE", + "authz": "broken object-level auth/IDOR, missing access checks, missing webhook/Slack signature verification, auth bypass", + "secrets-crypto": "hardcoded secrets/keys, weak/broken crypto (MD5/SHA1/unsalted), sensitive data in logs/errors/responses, wrong SSM-vs-Secrets-Manager placement", + "iac-iam": "wildcard IAM actions/resources, public buckets/endpoints, open security-group ingress (0.0.0.0/0), missing encryption, over-broad trust policies", + "web-client": "XSS (incl. dangerouslySetInnerHTML), CSRF, open redirect, client-side secret exposure", + "logic": "broken multi-step invariants, race conditions, missing tenant isolation, auth-state confusion", +} + +DETECTOR_TMPL = """You are a hostile {category} security auditor for Sea Haven. Assume this code is \ +hostile and the author missed something. Audit ONLY {category} issues in: {scope}. The repository root \ +is your current working directory; read the actual files with your tools. Ignore .git and any file that \ +is obviously an answer key. + +For each checklist item, either name a specific line that is provably safe, OR file a finding. Do not \ +hand-wave or give an open "looks fine" verdict. + +Checklist: {checklist} + +Return ONLY a JSON array (no prose, no markdown fences) where each element is: +{{"id": "", "title": "...", "claimed_severity": "critical|high|medium|low|info", + "cwe": "CWE-####", "file": "", "line": , "category": "{category}", + "data_flow": "numbered source->sink trace", "proof": {{"input": "concrete malicious input/trigger", + "outcome": "the specific bad result", "test": "optional failing-test sketch or null"}}, + "recommendation": "..."}} +If there are no findings, return [].""" + +VERIFIER_TMPL = """You are a skeptical exploitation verifier. You did NOT find these; your job is to \ +REFUTE weak claims. The repository root is your current working directory; read the real files to check \ +reachability before ruling. + +For each candidate finding decide whether there is a concrete, plausible proof-of-exploit (a specific \ +malicious input and the specific bad outcome, consistent with the code): +- If yes: set "status":"confirmed", keep "severity" equal to the claimed_severity, and tighten the proof. +- If no / speculative / not reachable: set "status":"unverified" and "severity":"unverified". Default to \ +unverified when uncertain. A confident assertion with no demonstrable input is NOT proof. + +Return ONLY a JSON array (no prose, no fences) of the SAME findings, each preserving its original "id", \ +"title", "cwe", "file", "line", "category", "claimed_severity", "data_flow", "recommendation", and adding \ +"severity" (final) plus "status" and the verified "proof". Return one element per candidate — do not drop \ +any. + +Candidates: +{candidates}""" + +SEV_RANK = { + "critical": 4, + "high": 3, + "medium": 2, + "low": 1, + "info": 0, + "unverified": -1, +} + + +def log(msg: str) -> None: + print(f"[run_headless] {msg}", file=sys.stderr, flush=True) + + +def extract_json(text: str): + """Best-effort parse of a JSON array/object from an agent's final text.""" + if not text: + return None + t = text.strip() + if t.startswith("```"): + t = re.sub(r"^```[a-zA-Z0-9]*\n?", "", t) + t = re.sub(r"\n?```\s*$", "", t).strip() + try: + return json.loads(t) + except Exception: + pass + # Fall back to the outermost [...] (or {...}) span. + for open_c, close_c in (("[", "]"), ("{", "}")): + start, end = t.find(open_c), t.rfind(close_c) + if 0 <= start < end: + try: + return json.loads(t[start : end + 1]) + except Exception: + continue + return None + + +async def run_agent( + prompt: str, *, cwd: Path, model: str | None, max_turns: int, budget_usd: float +) -> tuple[str, float, bool]: + """Run one fresh-context agent turn; return (final_text, cost_usd, is_error).""" + opts = ClaudeAgentOptions( + allowed_tools=READONLY_TOOLS, + disallowed_tools=BLOCKED_TOOLS, + permission_mode="bypassPermissions", + setting_sources=[], # hermetic: ignore user/project/local config + CLAUDE.md + cwd=str(cwd), + model=model, + max_turns=max_turns, + max_budget_usd=budget_usd, + ) + texts: list[str] = [] + result_text: str | None = None + cost = 0.0 + is_error = False + async for msg in query(prompt=prompt, options=opts): + name = type(msg).__name__ + if name == "AssistantMessage": + for block in getattr(msg, "content", []) or []: + t = getattr(block, "text", None) + if t: + texts.append(t) + elif name == "ResultMessage": + result_text = getattr(msg, "result", None) + cost = float(getattr(msg, "total_cost_usd", 0.0) or 0.0) + is_error = bool(getattr(msg, "is_error", False)) + return (result_text or "\n".join(texts)), cost, is_error + + +class Budget: + def __init__(self, total: float) -> None: + self.total = total + self.spent = 0.0 + self._lock = asyncio.Lock() + + async def add(self, amount: float) -> None: + async with self._lock: + self.spent += amount + + def exhausted(self) -> bool: + return self.total > 0 and self.spent >= self.total + + +async def run_detector( + category: str, + scope: str, + *, + cwd: Path, + model: str | None, + max_turns: int, + budget_usd: float, + sem: asyncio.Semaphore, + budget: Budget, +) -> tuple[str, list[dict], str | None]: + """Returns (category, candidate_findings, error_message_or_None).""" + async with sem: + if budget.exhausted(): + return category, [], "skipped: total budget exhausted" + prompt = DETECTOR_TMPL.format( + category=category, scope=scope, checklist=DETECTORS[category] + ) + t0 = time.monotonic() + try: + text, cost, is_error = await run_agent( + prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd + ) + except Exception as exc: # never let one detector kill the run + log(f"detector {category}: EXCEPTION {type(exc).__name__}: {exc}") + return category, [], f"exception: {type(exc).__name__}: {exc}" + await budget.add(cost) + dt = time.monotonic() - t0 + parsed = extract_json(text) + if not isinstance(parsed, list): + log( + f"detector {category}: UNPARSEABLE output ({dt:.0f}s, ${cost:.3f}) — over-reporting as error" + ) + return category, [], "unparseable detector output (NOT treated as clean)" + for f in parsed: + if isinstance(f, dict): + f.setdefault("category", category) + log( + f"detector {category}: {len(parsed)} candidate(s) ({dt:.0f}s, ${cost:.3f})" + + (" [is_error]" if is_error else "") + ) + return category, [f for f in parsed if isinstance(f, dict)], None + + +async def run_verifier( + candidates: list[dict], + *, + cwd: Path, + model: str | None, + max_turns: int, + budget_usd: float, + budget: Budget, +) -> tuple[dict[str, dict], float, str | None]: + """Returns ({id -> verdict}, cost, error). Empty verdicts on failure (caller keeps candidates).""" + prompt = VERIFIER_TMPL.format(candidates=json.dumps(candidates, indent=2)) + try: + text, cost, _ = await run_agent( + prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd + ) + except Exception as exc: + log(f"verifier: EXCEPTION {type(exc).__name__}: {exc}") + return {}, 0.0, f"exception: {type(exc).__name__}: {exc}" + await budget.add(cost) + parsed = extract_json(text) + if not isinstance(parsed, list): + log( + f"verifier: UNPARSEABLE output (${cost:.3f}) — keeping all candidates as unverified" + ) + return {}, cost, "unparseable verifier output" + verdicts = {f["id"]: f for f in parsed if isinstance(f, dict) and f.get("id")} + log(f"verifier: ruled on {len(verdicts)} finding(s) (${cost:.3f})") + return verdicts, cost, None + + +def merge(candidates: list[dict], verdicts: dict[str, dict]) -> list[dict]: + """Apply verifier verdicts to candidates. A candidate the verifier dropped or never + ruled on stays as 'unverified' — fail toward over-reporting, never silently delete.""" + out: list[dict] = [] + for c in candidates: + fid = c.get("id") or f"anon-{c.get('file', '?')}-{c.get('line', '?')}" + c.setdefault("id", fid) + v = verdicts.get(fid, {}) + status = v.get("status") + if status not in ("confirmed", "unverified", "suppressed"): + status = "unverified" + if status == "confirmed": + severity = v.get("severity") or c.get("claimed_severity") or "high" + else: + severity = "unverified" + out.append( + { + "id": fid, + "title": v.get("title") or c.get("title") or fid, + "severity": severity, + "claimed_severity": c.get("claimed_severity") or "high", + "cwe": c.get("cwe") or v.get("cwe") or "n/a", + "file": c.get("file") or v.get("file") or "", + "line": c.get("line", v.get("line")), + "category": c.get("category") or v.get("category") or "other", + "data_flow": v.get("data_flow") or c.get("data_flow") or "", + "proof": v.get("proof") + or c.get("proof") + or {"input": "", "outcome": ""}, + "status": status, + "recommendation": v.get("recommendation") + or c.get("recommendation") + or "", + } + ) + return out + + +def summarize(findings: list[dict]) -> dict: + conf_crit = sum( + 1 + for f in findings + if f["status"] == "confirmed" and f["severity"] == "critical" + ) + conf_high = sum( + 1 for f in findings if f["status"] == "confirmed" and f["severity"] == "high" + ) + return { + "confirmed_critical": conf_crit, + "confirmed_high": conf_high, + "block": (conf_crit + conf_high) > 0, + } + + +async def main_async(args: argparse.Namespace) -> int: + target = Path(args.target).resolve() + if not target.is_dir(): + log(f"target dir not found: {target}") + return 2 + scope = args.scope.strip() if args.scope else "the entire repository" + selected = ( + [d.strip() for d in args.detectors.split(",") if d.strip()] + if args.detectors + else list(DETECTORS) + ) + unknown = [d for d in selected if d not in DETECTORS] + if unknown: + log(f"unknown detector(s): {unknown}; valid: {list(DETECTORS)}") + return 2 + + budget = Budget(args.total_budget_usd) + sem = asyncio.Semaphore(max(1, args.concurrency)) + log( + f"target={target} scope='{scope}' detectors={selected} model={args.model or 'cli-default'} " + f"concurrency={args.concurrency} total_budget=${args.total_budget_usd}" + ) + + det_results = await asyncio.gather( + *[ + run_detector( + cat, + scope, + cwd=target, + model=args.model, + max_turns=args.max_turns, + budget_usd=args.detector_budget_usd, + sem=sem, + budget=budget, + ) + for cat in selected + ] + ) + + candidates: list[dict] = [] + errors: list[str] = [] + for cat, found, err in det_results: + candidates.extend(found) + if err: + errors.append(f"{cat}: {err}") + + log( + f"total candidates: {len(candidates)}; detector spend so far: ${budget.spent:.3f}" + ) + + if candidates and not budget.exhausted(): + verdicts, _, verr = await run_verifier( + candidates, + cwd=target, + model=args.model, + max_turns=args.max_turns, + budget_usd=args.detector_budget_usd, + budget=budget, + ) + if verr: + errors.append(f"verifier: {verr}") + else: + verdicts = {} + if budget.exhausted(): + errors.append( + "verifier: skipped (budget exhausted) — all candidates left unverified" + ) + + findings = merge(candidates, verdicts) + report = { + "findings": findings, + "summary": summarize(findings), + "_meta": { + "target": str(target), + "scope": scope, + "detectors": selected, + "model": args.model or "cli-default", + "spend_usd": round(budget.spent, 4), + "errors": errors, + }, + } + out = json.dumps(report, indent=2) + if args.out: + Path(args.out).write_text(out) + log(f"wrote {args.out}") + else: + print(out) + + s = report["summary"] + log( + f"DONE: {s['confirmed_critical']} confirmed-crit, {s['confirmed_high']} confirmed-high, " + f"block={s['block']}, spend=${budget.spent:.3f}, errors={len(errors)}" + ) + if errors: + for e in errors: + log(f" ERROR/NOTE: {e}") + return 0 + + +def main() -> None: + p = argparse.ArgumentParser( + description="Headless Sea Haven security detector fan-out + verifier" + ) + p.add_argument("target", help="target repository directory") + p.add_argument( + "--scope", default="", help="space-separated subdirs to restrict the audit" + ) + p.add_argument( + "--out", default="", help="write findings JSON here (default: stdout)" + ) + p.add_argument("--model", default=None, help="model id (default: CLI default)") + p.add_argument( + "--detectors", + default="", + help="comma list to restrict detectors (default: all 6)", + ) + p.add_argument( + "--concurrency", type=int, default=3, help="max concurrent detectors" + ) + p.add_argument( + "--max-turns", + type=int, + default=40, + help="max agent turns per detector/verifier", + ) + p.add_argument( + "--detector-budget-usd", type=float, default=2.0, help="per-call SDK spend cap" + ) + p.add_argument( + "--total-budget-usd", + type=float, + default=12.0, + help="overall spend cap (0 = unlimited)", + ) + args = p.parse_args() + + # Guarantee the subscription OAuth path: a raw API key would silently win, so remove it. + if os.environ.pop("ANTHROPIC_API_KEY", None): + log("removed ANTHROPIC_API_KEY from env to force the subscription OAuth path") + if not os.environ.get("CLAUDE_CODE_OAUTH_TOKEN"): + log( + "FATAL: CLAUDE_CODE_OAUTH_TOKEN not set (source ~/secrev.env). Refusing to run." + ) + sys.exit(2) + + sys.exit(asyncio.run(main_async(args))) + + +if __name__ == "__main__": + main() -- 2.50.1 From c217c5656d8e9915eb74aa547210384a26fa8444 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:33:43 -0400 Subject: [PATCH 07/16] Add machine-level suppressions to the repo-sourced hooks The live global pre-push hook was hand-edited to resolve suppressions from a machine-level file (${SH_SECURITY_SUPPRESSIONS_DIR:-~/.config/sea-haven/security-review}/ /suppressions.json) kept out of repo history, falling back to a repo-local .security-review/suppressions.json. The repo-sourced hooks lacked it, so install-hooks.sh --global would overwrite the live hook and lose the feature. Port the prefer-machine/fallback-repo-local block into hooks/pre-push, align hooks/pre-commit to the same (else a machine-suppressed finding passes at push but blocks at commit), and document the path + SH_SECURITY_SUPPRESSIONS_DIR override + basename-collision caveat in the README. --- security-review/README.md | 19 +++++++++++++++---- security-review/hooks/pre-commit | 11 ++++++++++- security-review/hooks/pre-push | 11 ++++++++++- 3 files changed, 35 insertions(+), 6 deletions(-) diff --git a/security-review/README.md b/security-review/README.md index dc23536..c4cc214 100644 --- a/security-review/README.md +++ b/security-review/README.md @@ -32,10 +32,21 @@ security-review/install-hooks.sh --global security-review/install-hooks.sh /path/to/repo ``` The global mode sets `git config --global core.hooksPath ~/.config/git/hooks`. Skip a repo with a -`.security-review-skip` file at its root; suppress a specific false positive in the repo's -`.security-review/suppressions.json` (a written justification is required and is surfaced); bypass once -with `git push --no-verify`. Caveat: a repo with its own local `core.hooksPath` overrides the global hook -— install per-repo there. See memory `reference_global_security_review_hook`. +`.security-review-skip` file at its root; bypass once with `git push --no-verify`. Caveat: a repo with +its own local `core.hooksPath` overrides the global hook — install per-repo there. See memory +`reference_global_security_review_hook`. + +**Suppressing a false positive.** A written justification is required and is surfaced in the report. The +hooks resolve a suppressions file in this order: +1. **Machine-level (preferred), kept out of repo history:** + `${SH_SECURITY_SUPPRESSIONS_DIR:-~/.config/sea-haven/security-review}//suppressions.json` + (override the base dir with `SH_SECURITY_SUPPRESSIONS_DIR`). Keeps a suppression from becoming a + permanent in-history "ignore." +2. **Repo-local fallback:** `/.security-review/suppressions.json` (used only if no machine-level file exists). + +Same JSON either place: `{"suppressions":[{"id":"","justification":"…"}]}`. Caveat: +machine-level files are keyed by **repo basename**, so two repos sharing a name collide — fine for the +current single-namespace layout under `~/Documents/repositories`. ## No CI — by design There is **no CI** wiring for this gate. For a solo dev the git hooks + nightly VM sweep are the backstop, diff --git a/security-review/hooks/pre-commit b/security-review/hooks/pre-commit index c28b1f6..268973d 100755 --- a/security-review/hooks/pre-commit +++ b/security-review/hooks/pre-commit @@ -14,6 +14,15 @@ fi # Nothing staged -> nothing to do. git diff --cached --name-only --diff-filter=ACM | grep -q . || exit 0 SUP=() -[ -f "$REPO_ROOT/.security-review/suppressions.json" ] && SUP=(--suppressions "$REPO_ROOT/.security-review/suppressions.json") +# Suppressions: prefer a MACHINE-LEVEL file kept out of repo history +# (//suppressions.json), else fall back to a repo-local +# .security-review/suppressions.json. Keyed by repo basename — adequate for the +# current single-namespace layout under ~/Documents/repositories. +MACHINE_SUP="${SH_SECURITY_SUPPRESSIONS_DIR:-$HOME/.config/sea-haven/security-review}/$(basename "$REPO_ROOT")/suppressions.json" +if [ -f "$MACHINE_SUP" ]; then + SUP=(--suppressions "$MACHINE_SUP") +elif [ -f "$REPO_ROOT/.security-review/suppressions.json" ]; then + SUP=(--suppressions "$REPO_ROOT/.security-review/suppressions.json") +fi # ${SUP[@]+"${SUP[@]}"} = bash-3.2-safe expansion of a possibly-empty array under set -u. exec bash "$REVIEW_SH" --scanners-only ${SUP[@]+"${SUP[@]}"} "$REPO_ROOT" diff --git a/security-review/hooks/pre-push b/security-review/hooks/pre-push index 74118d8..7ca47df 100755 --- a/security-review/hooks/pre-push +++ b/security-review/hooks/pre-push @@ -10,7 +10,16 @@ REPO_ROOT="$(git rev-parse --show-toplevel 2>/dev/null)" || exit 0 REVIEW_SH="${SH_REVIEW_SH:-$HOME/Documents/repositories/orchestrator/security-review/review.sh}" if [ -f "$REVIEW_SH" ]; then SUP=() - [ -f "$REPO_ROOT/.security-review/suppressions.json" ] && SUP=(--suppressions "$REPO_ROOT/.security-review/suppressions.json") + # Suppressions: prefer a MACHINE-LEVEL file kept out of repo history + # (//suppressions.json), else fall back to a repo-local + # .security-review/suppressions.json. Keyed by repo basename — adequate for the + # current single-namespace layout under ~/Documents/repositories. + MACHINE_SUP="${SH_SECURITY_SUPPRESSIONS_DIR:-$HOME/.config/sea-haven/security-review}/$(basename "$REPO_ROOT")/suppressions.json" + if [ -f "$MACHINE_SUP" ]; then + SUP=(--suppressions "$MACHINE_SUP") + elif [ -f "$REPO_ROOT/.security-review/suppressions.json" ]; then + SUP=(--suppressions "$REPO_ROOT/.security-review/suppressions.json") + fi echo "security-review: scanning $REPO_ROOT (scanners-only) before push..." >&2 # ${SUP[@]+"${SUP[@]}"} = bash-3.2-safe expansion of a possibly-empty array under set -u. if ! bash "$REVIEW_SH" --scanners-only ${SUP[@]+"${SUP[@]}"} "$REPO_ROOT"; then -- 2.50.1 From 80d539178589f052abc97335f1efee47f934992c Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:55:01 -0400 Subject: [PATCH 08/16] Document the security-review subsystem in the root README The root README covered only the router and omitted the security-review/ subsystem entirely (Path A skill, review.sh gate, Path B nightly sweep, global hooks). Add a Security Review section summarizing both paths and linking the subsystem README + DEPLOY-R720 runbook. --- README.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/README.md b/README.md index bdc29bd..3f282c3 100644 --- a/README.md +++ b/README.md @@ -113,6 +113,16 @@ All connections authenticated under Composio user `amoussa`: The connector node is restricted to **one tool call per run** — a load-bearing rule learned from a 1.9M-token incident with meta-tool routing. +## Security Review + +The `security-review/` subsystem is a high-recall, anti-complacency security gate. It is **separate from the router** — it does not route through `run.py` or LangGraph. One pure-code script, `review.sh`, owns the **block decision** (confirmed critical/high → block); no agent decides. + +- **Path A — interactive:** the `/sh-security-review` Claude Code skill (Max-covered). Narrow fresh-context detector fan-out + a proof-or-kill verifier; emits the structured finding schema for `review.sh` to gate. +- **Path B — unattended:** a nightly two-tier sweep on the `sh-secrev` R720 VM. Tier 1 runs deterministic scanners (`review.sh --scanners-only`) over every Sea-Haven-Industries org repo; Tier 2 is a budget-bounded agentic pass (`run_headless.py`) on a round-robin rotation. Clean-clone auto-discovery via a read-only GitHub PAT; **ALARM-only** Slack (a clean night posts nothing). +- **Git hooks:** global pre-commit / pre-push hooks (`install-hooks.sh --global`) gate every local repo via `review.sh --scanners-only`. + +See `security-review/README.md` for full detail and `security-review/DEPLOY-R720.md` for the VM runbook. + ## Setup 1. Install dependencies: `pip install -r requirements.txt` -- 2.50.1 From 572ae3ca416c9c315bfdd91a4d4837ec07e01b61 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:55:01 -0400 Subject: [PATCH 09/16] Move orchestrator gitleaks suppression out of repo history Per the machine-level-suppressions policy, the .env.example FP suppression now lives at ~/.config/sea-haven/security-review/orchestrator/suppressions.json (resolved first by the hooks), not a committed .security-review/suppressions.json. Gate still passes via the machine-level file. --- .security-review/suppressions.json | 8 -------- 1 file changed, 8 deletions(-) delete mode 100644 .security-review/suppressions.json diff --git a/.security-review/suppressions.json b/.security-review/suppressions.json deleted file mode 100644 index 56cee62..0000000 --- a/.security-review/suppressions.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "suppressions": [ - { - "id": "gitleaks-generic-api-key-2", - "justification": "False positive. .env.example:2 is a documented placeholder token (not a live secret) that exists to show the required env var shape. gitleaks runs in git-mode and scans committed history, so it flags the placeholder even though the working-tree value is inert. No real credential is or was exposed. Reviewed 2026-06-16." - } - ] -} -- 2.50.1 From 028c844c8665dbc92ccf7e7368289e5e44de6c1f Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:55:01 -0400 Subject: [PATCH 10/16] Move the May handoff doc under docs/ De-clutter the repo root; HANDOFF-2026-05-15.md now lives in docs/. --- HANDOFF-2026-05-15.md => docs/HANDOFF-2026-05-15.md | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename HANDOFF-2026-05-15.md => docs/HANDOFF-2026-05-15.md (100%) diff --git a/HANDOFF-2026-05-15.md b/docs/HANDOFF-2026-05-15.md similarity index 100% rename from HANDOFF-2026-05-15.md rename to docs/HANDOFF-2026-05-15.md -- 2.50.1 From 86ad29dcc29811c8a2c9b13e22be20064926fffb Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 15:55:01 -0400 Subject: [PATCH 11/16] Pin direct dependencies to exact versions requirements.txt used loose >= ranges; pin the 7 direct deps to the known-good installed versions (Python 3.12.13) for a reproducible venv. --- requirements.txt | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/requirements.txt b/requirements.txt index 97aceac..b4c01cf 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,7 +1,7 @@ -langgraph>=1.1.0 -langchain-anthropic>=1.4.0 -langchain-openai>=1.2.0 -langchain-google-genai>=4.2.0 -langchain-community>=0.4.0 -composio-langgraph>=0.13.0 -python-dotenv>=1.0.0 +langgraph==1.1.10 +langchain-anthropic==1.4.3 +langchain-openai==1.2.1 +langchain-google-genai==4.2.2 +langchain-community==0.4.1 +composio-langgraph==0.13.0 +python-dotenv==1.2.2 -- 2.50.1 From 6d65c54b58b9f853f78568c1503e5c58125d3295 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 16:00:21 -0400 Subject: [PATCH 12/16] Fix unbound-variable crash in the nightly sweep tier-1 loop scan_scanners referenced ${slug} inside the SAME local statement that defines it (local target=... slug=$2 result_json=...${slug}...). bash expands the local's arguments before the builtin assigns them, so under set -u ${slug} is unbound and the sweep died right after the canary, before tier 1 ever ran. Split the local so slug exists first. Fix the same latent self-reference in mirror_repo (dir=...$name), which only worked by accident because the discovery loop left a global $name. --- security-review/nightly_sweep.sh | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/security-review/nightly_sweep.sh b/security-review/nightly_sweep.sh index 999ee3d..fefa4f2 100755 --- a/security-review/nightly_sweep.sh +++ b/security-review/nightly_sweep.sh @@ -136,7 +136,8 @@ discover_repos() { # failed fetch cannot leave GH_TOKEN at rest on disk. (Residual: the token is briefly visible # in process args to a local `ps`; acceptable on this single-user unattended box.) mirror_repo() { # name clone_url default_branch -> 0 ok / 1 fail - local name="$1" url="$2" branch="$3" dir="$MIRROR_DIR/$name" + local name="$1" url="$2" branch="$3" + local dir="$MIRROR_DIR/$name" local auth_url="https://x-access-token:${GH_TOKEN}@${url#https://}" if [ -d "$dir/.git" ]; then git -C "$dir" fetch --depth=1 "$auth_url" "$branch" >/dev/null 2>&1 || return 1 @@ -188,7 +189,8 @@ xmodel_check_criticals() { # Sets T1_BLOCK/T1_CRIT/T1_HIGH. review.sh exit 0 pass / 1 BLOCK / 2 setup. T1_BLOCK=0; T1_CRIT=0; T1_HIGH=0 scan_scanners() { # target slug - local target="$1" slug="$2" result_json="$REPORT_DIR/${slug}.scanners.json" + local target="$1" slug="$2" + local result_json="$REPORT_DIR/${slug}.scanners.json" T1_BLOCK=0; T1_CRIT=0; T1_HIGH=0 local sup=() [ -f "$target/.security-review/suppressions.json" ] && sup=(--suppressions "$target/.security-review/suppressions.json") -- 2.50.1 From f4bb2bce8af968c1b99bad872bff299bcbfa8540 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 16:43:35 -0400 Subject: [PATCH 13/16] Make scan_scanners return 0 so a passing repo doesn't trip set -e scan_scanners communicates results via globals (T1_BLOCK/T1_CRIT/T1_HIGH); its last statement was a bare [ $rc -eq 1 ] && T1_BLOCK=1. On a PASS (rc=0) that test is false, so the function returned non-zero and set -e killed the whole sweep at the first passing repo in tier 1 (right after the unbound-variable fix let it get that far). Add an explicit return 0. Audited the rest of the tier1/tier2/summary path; scan_agentic and the others already end on a zero-status command. --- security-review/nightly_sweep.sh | 1 + 1 file changed, 1 insertion(+) diff --git a/security-review/nightly_sweep.sh b/security-review/nightly_sweep.sh index fefa4f2..a0b88ab 100755 --- a/security-review/nightly_sweep.sh +++ b/security-review/nightly_sweep.sh @@ -202,6 +202,7 @@ scan_scanners() { # target slug T1_CRIT="$(jq -r '(.summary.confirmed_critical // 0)' "$result_json" 2>/dev/null || echo 0)" T1_HIGH="$(jq -r '(.summary.confirmed_high // 0)' "$result_json" 2>/dev/null || echo 0)" [ "$rc" -eq 1 ] && T1_BLOCK=1 + return 0 # MUST return 0: results go via globals; a falsey last cmd would trip set -e in the caller } # --- TIER 2: agentic run_headless + full review.sh over a target dir ---------- -- 2.50.1 From 15a16f8b88fa42586c53f9ef9c7ce2630c2da092 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Tue, 16 Jun 2026 18:02:57 -0400 Subject: [PATCH 14/16] Tune nightly defaults from first-run data First full VM run: canary recall 18 with the expanded Node/.NET corpus, and a real agentic repo cost ~$3.50 (vs the $1.4 testbed). Raise CANARY_FLOOR 10->14 (catches a language-blindness recall collapse to ~9 while keeping margin under 18) and MAX_CYCLE_NIGHTS 4->6 (at ~4-6 agentic repos/night a full 22-repo rotation takes ~4-5 nights; 4 would false-fire the coverage alarm). Both stay env-overridable and re-tunable as data accrues. --- security-review/nightly_sweep.sh | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/security-review/nightly_sweep.sh b/security-review/nightly_sweep.sh index a0b88ab..cab297a 100755 --- a/security-review/nightly_sweep.sh +++ b/security-review/nightly_sweep.sh @@ -67,10 +67,10 @@ GH_ORG="${GH_ORG:-Sea-Haven-Industries}" MIRROR_DIR="${MIRROR_DIR:-$HOME/repo-mirrors}" CENTRAL_SKIP_FILE="${CENTRAL_SKIP_FILE:-$HOME/.secrev-skip.txt}" TESTBED="${TESTBED:-$HOME/security-review-testbed}" -CANARY_FLOOR="${CANARY_FLOOR:-10}" +CANARY_FLOOR="${CANARY_FLOOR:-14}" TOTAL_BUDGET_USD="${TOTAL_BUDGET_USD:-20}" PER_TARGET_BUDGET_USD="${PER_TARGET_BUDGET_USD:-12}" -MAX_CYCLE_NIGHTS="${MAX_CYCLE_NIGHTS:-4}" +MAX_CYCLE_NIGHTS="${MAX_CYCLE_NIGHTS:-6}" MAX_AGENTIC_PER_NIGHT="${MAX_AGENTIC_PER_NIGHT:-0}" REPORT_ROOT="${REPORT_ROOT:-$HOME/sweep-reports}" ENABLE_XMODEL_HOOK="${ENABLE_XMODEL_HOOK:-0}" -- 2.50.1 From dbe2f8c186d55e1289029fc154270c3c2e40d5cb Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Wed, 17 Jun 2026 13:18:46 -0400 Subject: [PATCH 15/16] Raise nightly agentic budget $20 -> $120 for full per-night coverage First-run data (2026-06-17) showed the $20 ceiling covered only the canary + 5 of ~22 scannable repos before pausing the rotation, leaving 16 repos un-deep-scanned that night. Raise TOTAL_BUDGET_USD default to $120 so every repo gets a deep agentic pass each night (~22 x ~$5 + canary, with headroom). Spend draws on the Max subscription pool; the per-target cap ($12) and round-robin rotation are unchanged, so this is a ceiling raise, not a per-repo cost change. Updates README, DEPLOY, and the systemd Environment example to match. --- security-review/DEPLOY-R720.md | 2 +- security-review/README.md | 2 +- security-review/nightly_sweep.sh | 6 ++++-- security-review/systemd/sea-haven-secrev.service | 2 +- 4 files changed, 7 insertions(+), 5 deletions(-) diff --git a/security-review/DEPLOY-R720.md b/security-review/DEPLOY-R720.md index 4f1259b..0788685 100644 --- a/security-review/DEPLOY-R720.md +++ b/security-review/DEPLOY-R720.md @@ -101,7 +101,7 @@ ALARM-only (a clean night posts nothing). Secret-shaped values are redacted from under `~/sweep-reports//` are mode 600. ### Config (env / systemd `Environment=`) -`GH_ORG` (Sea-Haven-Industries) · `MIRROR_DIR` (~/repo-mirrors) · `TOTAL_BUDGET_USD` (20) · +`GH_ORG` (Sea-Haven-Industries) · `MIRROR_DIR` (~/repo-mirrors) · `TOTAL_BUDGET_USD` (120) · `PER_TARGET_BUDGET_USD` (12) · `CANARY_FLOOR` (10) · `MAX_CYCLE_NIGHTS` (4) · `MAX_AGENTIC_PER_NIGHT` (0 = unlimited) · `CENTRAL_SKIP_FILE` (~/.secrev-skip.txt) · `ENABLE_XMODEL_HOOK` (0) · `TARGETS` (manual override — scan explicit paths, no discovery). diff --git a/security-review/README.md b/security-review/README.md index c4cc214..f68f625 100644 --- a/security-review/README.md +++ b/security-review/README.md @@ -93,7 +93,7 @@ It is **ALARM-only**: a clean night posts nothing. See memory `feedback_cloudwat - **Canary check:** the testbed MUST block AND surface ≥ `CANARY_FLOOR` (default 10) confirmed crit/high. Otherwise → COMPLACENCY ALARM. The corpus now includes Node + .NET fixtures (see the testbed key); re-tune the floor after the first VM canary run reports the expanded recall number. -- **Budget ceiling:** `TOTAL_BUDGET_USD` (default 20) caps aggregate agentic spend; `PER_TARGET_BUDGET_USD` +- **Budget ceiling:** `TOTAL_BUDGET_USD` (default 120 — full deep-pass coverage of every repo per night) caps aggregate agentic spend; `PER_TARGET_BUDGET_USD` (default 12) caps each repo; `MAX_AGENTIC_PER_NIGHT` (default 0 = unlimited) optionally caps wall-clock. With the SDK-billing split deferred (memory `reference-claude-subscription-billing`), spend draws from the Max subscription limits, so the two-tier design keeps full coverage cheap and bounds the agentic draw. diff --git a/security-review/nightly_sweep.sh b/security-review/nightly_sweep.sh index cab297a..5a4634d 100755 --- a/security-review/nightly_sweep.sh +++ b/security-review/nightly_sweep.sh @@ -39,7 +39,9 @@ # TARGETS space-separated paths to scan INSTEAD of discovery (manual override) # TESTBED canary corpus dir (default: ~/security-review-testbed) # CANARY_FLOOR min confirmed crit+high the canary MUST surface (default: 10) -# TOTAL_BUDGET_USD hard agentic spend ceiling across the night (default: 20) +# TOTAL_BUDGET_USD hard agentic spend ceiling across the night (default: 120 — +# full deep-pass coverage of every repo per night; first-run +# data 2026-06-17 showed $20 covered only canary + 5 repos) # PER_TARGET_BUDGET_USD passed to run_headless --total-budget-usd (default: 12) # MAX_CYCLE_NIGHTS alarm if the agentic rotation hasn't covered every repo in this many nights (default: 4) # MAX_AGENTIC_PER_NIGHT cap on repos given the deep agentic pass per night, for wall-clock bounding @@ -68,7 +70,7 @@ MIRROR_DIR="${MIRROR_DIR:-$HOME/repo-mirrors}" CENTRAL_SKIP_FILE="${CENTRAL_SKIP_FILE:-$HOME/.secrev-skip.txt}" TESTBED="${TESTBED:-$HOME/security-review-testbed}" CANARY_FLOOR="${CANARY_FLOOR:-14}" -TOTAL_BUDGET_USD="${TOTAL_BUDGET_USD:-20}" +TOTAL_BUDGET_USD="${TOTAL_BUDGET_USD:-120}" PER_TARGET_BUDGET_USD="${PER_TARGET_BUDGET_USD:-12}" MAX_CYCLE_NIGHTS="${MAX_CYCLE_NIGHTS:-6}" MAX_AGENTIC_PER_NIGHT="${MAX_AGENTIC_PER_NIGHT:-0}" diff --git a/security-review/systemd/sea-haven-secrev.service b/security-review/systemd/sea-haven-secrev.service index 3e40fb0..cb53527 100644 --- a/security-review/systemd/sea-haven-secrev.service +++ b/security-review/systemd/sea-haven-secrev.service @@ -31,7 +31,7 @@ WorkingDirectory=/home/adam/orchestrator EnvironmentFile=-/home/adam/secrev.env EnvironmentFile=-/home/adam/orchestrator/.env # Tune ceilings/targets here without editing the script (uncomment to override defaults): -# Environment=TOTAL_BUDGET_USD=20 +# Environment=TOTAL_BUDGET_USD=120 # Environment=PER_TARGET_BUDGET_USD=12 # Environment=CANARY_FLOOR=10 # Environment=MAX_CYCLE_NIGHTS=4 -- 2.50.1 From c6ecc621bea9f91219000b23840438530015acb7 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Wed, 17 Jun 2026 14:55:41 -0400 Subject: [PATCH 16/16] Prune generated/vendored trees from the scanners cfn-lint, semgrep, and checkov were scanning synthesized/vendored output (cdk.out, node_modules, .venv/venv, .aws-sam, dist, build). On CDK repos this explodes the find/xargs arg list and stalls the scan, and flagging synthesized templates is wrong. Prune those trees in the cfn-lint find, and pass --exclude / --skip-path to semgrep / checkov. --- security-review/review.sh | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/security-review/review.sh b/security-review/review.sh index af5f766..1980449 100755 --- a/security-review/review.sh +++ b/security-review/review.sh @@ -43,7 +43,9 @@ note_missing() { echo " [MISSING] $1 — not run. Install: $2" >&2; } # --- cfn-lint (installed): lint SAM/CFN templates. Normalize to findings. --- if command -v cfn-lint >/dev/null; then - TPLS="$(scope_paths | xargs -I{} find {} -type f \( -name '*.yaml' -o -name '*.yml' \) 2>/dev/null \ + # Prune generated/vendored trees (cdk.out, node_modules, …): scanning synthesized output is + # wrong and, on CDK repos, explodes the arg list / stalls the scanners. + TPLS="$(scope_paths | xargs -I{} find {} \( -type d \( -name cdk.out -o -name node_modules -o -name .git -o -name .aws-sam -o -name .venv -o -name venv -o -name dist -o -name build \) -prune \) -o \( -type f \( -name '*.yaml' -o -name '*.yml' \) -print \) 2>/dev/null \ | xargs -I{} sh -c 'grep -lE "AWSTemplateFormatVersion|Transform: *AWS::Serverless" "{}" 2>/dev/null || true')" if [ -n "$TPLS" ]; then # shellcheck disable=SC2086 @@ -67,7 +69,7 @@ SCOPE_PATHS="$(scope_paths)" # --- semgrep (SAST: injection/authz/xss/secrets) --- if command -v semgrep >/dev/null; then # shellcheck disable=SC2086 - if SG="$(semgrep --config p/security-audit --config p/secrets --config p/javascript --json --metrics=off $SCOPE_PATHS 2>/dev/null)"; then + if SG="$(semgrep --config p/security-audit --config p/secrets --config p/javascript --json --metrics=off --exclude cdk.out --exclude node_modules --exclude .venv --exclude venv --exclude .aws-sam --exclude dist --exclude build $SCOPE_PATHS 2>/dev/null)"; then NORM="$(echo "$SG" | jq '[.results[] | { id: ("semgrep-" + (.check_id|split(".")|last) + "-" + (.start.line|tostring)), title: ((.check_id|split(".")|last) + ": " + ((.extra.message // "")[0:120])), @@ -109,7 +111,7 @@ else note_missing gitleaks "brew install gitleaks"; fi if command -v checkov >/dev/null; then CKALL="$TMP/ck.json"; echo '[]' > "$CKALL" for p in $SCOPE_PATHS; do - if [ -d "$p" ]; then RAW="$(checkov -d "$p" -o json --compact --quiet 2>/dev/null || true)" + if [ -d "$p" ]; then RAW="$(checkov -d "$p" --skip-path cdk.out --skip-path node_modules --skip-path .venv --skip-path venv --skip-path .aws-sam --skip-path dist --skip-path build -o json --compact --quiet 2>/dev/null || true)" else RAW="$(checkov -f "$p" -o json --compact --quiet 2>/dev/null || true)"; fi [ -z "$RAW" ] && continue FC="$(echo "$RAW" | jq '[ (if type=="array" then .[] else . end).results.failed_checks // [] ] | add // []' 2>/dev/null || echo '[]')" -- 2.50.1