#!/usr/bin/env python3 """run_headless.py — Path B headless detector fan-out + proof-or-kill verifier. Reuses the /sh-security-review detector + verifier prompts, but runs them unattended via the Claude Agent SDK instead of interactive Claude Code subagents. Authenticates with the Claude subscription OAuth token (CLAUDE_CODE_OAUTH_TOKEN) through the bundled `claude` CLI — NEVER a raw ANTHROPIC_API_KEY (which would be metered and would silently win if both were set, so we pop it). Emits the finding-schema JSON ({findings, summary}) that `review.sh --agent-findings` consumes. review.sh re-derives the gate, so this script's job is high-recall candidate generation + proof-or-kill verification, failing toward over-reporting (never silently drops a candidate or a parse error). See memory project-security-review-agent and security-review/DEPLOY-R720.md. Usage: CLAUDE_CODE_OAUTH_TOKEN=... python3 run_headless.py TARGET_DIR \ [--scope "src infra web"] [--out findings.json] [--model claude-...] \ [--concurrency 3] [--detectors injection,authz] \ [--detector-budget-usd 2.0] [--total-budget-usd 12.0] """ from __future__ import annotations import argparse import asyncio import json import os import re import sys import time from pathlib import Path from claude_agent_sdk import ClaudeAgentOptions, query # Read-only surface: detectors reason over source, they don't mutate or fetch. READONLY_TOOLS = ["Read", "Grep", "Glob"] BLOCKED_TOOLS = ["Bash", "Write", "Edit", "NotebookEdit", "WebFetch", "WebSearch"] # Detector category -> closed checklist (verbatim intent from sh-security-review.md). DETECTORS: dict[str, str] = { "injection": "SQL/command/template injection, unsafe deserialization, SSRF, path/file traversal, XXE", "authz": "broken object-level auth/IDOR, missing access checks, missing webhook/Slack signature verification, auth bypass", "secrets-crypto": "hardcoded secrets/keys, weak/broken crypto (MD5/SHA1/unsalted), sensitive data in logs/errors/responses, wrong SSM-vs-Secrets-Manager placement", "iac-iam": "wildcard IAM actions/resources, public buckets/endpoints, open security-group ingress (0.0.0.0/0), missing encryption, over-broad trust policies", "web-client": "XSS (incl. dangerouslySetInnerHTML), CSRF, open redirect, client-side secret exposure", "logic": "broken multi-step invariants, race conditions, missing tenant isolation, auth-state confusion", } DETECTOR_TMPL = """You are a hostile {category} security auditor for Sea Haven. Assume this code is \ hostile and the author missed something. Audit ONLY {category} issues in: {scope}. The repository root \ is your current working directory; read the actual files with your tools. Ignore .git and any file that \ is obviously an answer key. For each checklist item, either name a specific line that is provably safe, OR file a finding. Do not \ hand-wave or give an open "looks fine" verdict. Checklist: {checklist} Return ONLY a JSON array (no prose, no markdown fences) where each element is: {{"id": "", "title": "...", "claimed_severity": "critical|high|medium|low|info", "cwe": "CWE-####", "file": "", "line": , "category": "{category}", "data_flow": "numbered source->sink trace", "proof": {{"input": "concrete malicious input/trigger", "outcome": "the specific bad result", "test": "optional failing-test sketch or null"}}, "recommendation": "..."}} If there are no findings, return [].""" VERIFIER_TMPL = """You are a skeptical exploitation verifier. You did NOT find these; your job is to \ REFUTE weak claims. The repository root is your current working directory; read the real files to check \ reachability before ruling. For each candidate finding decide whether there is a concrete, plausible proof-of-exploit (a specific \ malicious input and the specific bad outcome, consistent with the code): - If yes: set "status":"confirmed", keep "severity" equal to the claimed_severity, and tighten the proof. - If no / speculative / not reachable: set "status":"unverified" and "severity":"unverified". Default to \ unverified when uncertain. A confident assertion with no demonstrable input is NOT proof. Return ONLY a JSON array (no prose, no fences) of the SAME findings, each preserving its original "id", \ "title", "cwe", "file", "line", "category", "claimed_severity", "data_flow", "recommendation", and adding \ "severity" (final) plus "status" and the verified "proof". Return one element per candidate — do not drop \ any. Candidates: {candidates}""" SEV_RANK = { "critical": 4, "high": 3, "medium": 2, "low": 1, "info": 0, "unverified": -1, } def log(msg: str) -> None: print(f"[run_headless] {msg}", file=sys.stderr, flush=True) def extract_json(text: str): """Best-effort parse of a JSON array/object from an agent's final text.""" if not text: return None t = text.strip() if t.startswith("```"): t = re.sub(r"^```[a-zA-Z0-9]*\n?", "", t) t = re.sub(r"\n?```\s*$", "", t).strip() try: return json.loads(t) except Exception: pass # Fall back to the outermost [...] (or {...}) span. for open_c, close_c in (("[", "]"), ("{", "}")): start, end = t.find(open_c), t.rfind(close_c) if 0 <= start < end: try: return json.loads(t[start : end + 1]) except Exception: continue return None async def run_agent( prompt: str, *, cwd: Path, model: str | None, max_turns: int, budget_usd: float ) -> tuple[str, float, bool]: """Run one fresh-context agent turn; return (final_text, cost_usd, is_error).""" opts = ClaudeAgentOptions( allowed_tools=READONLY_TOOLS, disallowed_tools=BLOCKED_TOOLS, permission_mode="bypassPermissions", setting_sources=[], # hermetic: ignore user/project/local config + CLAUDE.md cwd=str(cwd), model=model, max_turns=max_turns, max_budget_usd=budget_usd, ) texts: list[str] = [] result_text: str | None = None cost = 0.0 is_error = False async for msg in query(prompt=prompt, options=opts): name = type(msg).__name__ if name == "AssistantMessage": for block in getattr(msg, "content", []) or []: t = getattr(block, "text", None) if t: texts.append(t) elif name == "ResultMessage": result_text = getattr(msg, "result", None) cost = float(getattr(msg, "total_cost_usd", 0.0) or 0.0) is_error = bool(getattr(msg, "is_error", False)) return (result_text or "\n".join(texts)), cost, is_error class Budget: def __init__(self, total: float) -> None: self.total = total self.spent = 0.0 self._lock = asyncio.Lock() async def add(self, amount: float) -> None: async with self._lock: self.spent += amount def exhausted(self) -> bool: return self.total > 0 and self.spent >= self.total async def run_detector( category: str, scope: str, *, cwd: Path, model: str | None, max_turns: int, budget_usd: float, sem: asyncio.Semaphore, budget: Budget, ) -> tuple[str, list[dict], str | None]: """Returns (category, candidate_findings, error_message_or_None).""" async with sem: if budget.exhausted(): return category, [], "skipped: total budget exhausted" prompt = DETECTOR_TMPL.format( category=category, scope=scope, checklist=DETECTORS[category] ) t0 = time.monotonic() try: text, cost, is_error = await run_agent( prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd ) except Exception as exc: # never let one detector kill the run log(f"detector {category}: EXCEPTION {type(exc).__name__}: {exc}") return category, [], f"exception: {type(exc).__name__}: {exc}" await budget.add(cost) dt = time.monotonic() - t0 parsed = extract_json(text) if not isinstance(parsed, list): log( f"detector {category}: UNPARSEABLE output ({dt:.0f}s, ${cost:.3f}) — over-reporting as error" ) return category, [], "unparseable detector output (NOT treated as clean)" for f in parsed: if isinstance(f, dict): f.setdefault("category", category) log( f"detector {category}: {len(parsed)} candidate(s) ({dt:.0f}s, ${cost:.3f})" + (" [is_error]" if is_error else "") ) return category, [f for f in parsed if isinstance(f, dict)], None async def run_verifier( candidates: list[dict], *, cwd: Path, model: str | None, max_turns: int, budget_usd: float, budget: Budget, ) -> tuple[dict[str, dict], float, str | None]: """Returns ({id -> verdict}, cost, error). Empty verdicts on failure (caller keeps candidates).""" prompt = VERIFIER_TMPL.format(candidates=json.dumps(candidates, indent=2)) try: text, cost, _ = await run_agent( prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd ) except Exception as exc: log(f"verifier: EXCEPTION {type(exc).__name__}: {exc}") return {}, 0.0, f"exception: {type(exc).__name__}: {exc}" await budget.add(cost) parsed = extract_json(text) if not isinstance(parsed, list): log( f"verifier: UNPARSEABLE output (${cost:.3f}) — keeping all candidates as unverified" ) return {}, cost, "unparseable verifier output" verdicts = {f["id"]: f for f in parsed if isinstance(f, dict) and f.get("id")} log(f"verifier: ruled on {len(verdicts)} finding(s) (${cost:.3f})") return verdicts, cost, None def merge(candidates: list[dict], verdicts: dict[str, dict]) -> list[dict]: """Apply verifier verdicts to candidates. A candidate the verifier dropped or never ruled on stays as 'unverified' — fail toward over-reporting, never silently delete.""" out: list[dict] = [] for c in candidates: fid = c.get("id") or f"anon-{c.get('file', '?')}-{c.get('line', '?')}" c.setdefault("id", fid) v = verdicts.get(fid, {}) status = v.get("status") if status not in ("confirmed", "unverified", "suppressed"): status = "unverified" if status == "confirmed": severity = v.get("severity") or c.get("claimed_severity") or "high" else: severity = "unverified" out.append( { "id": fid, "title": v.get("title") or c.get("title") or fid, "severity": severity, "claimed_severity": c.get("claimed_severity") or "high", "cwe": c.get("cwe") or v.get("cwe") or "n/a", "file": c.get("file") or v.get("file") or "", "line": c.get("line", v.get("line")), "category": c.get("category") or v.get("category") or "other", "data_flow": v.get("data_flow") or c.get("data_flow") or "", "proof": v.get("proof") or c.get("proof") or {"input": "", "outcome": ""}, "status": status, "recommendation": v.get("recommendation") or c.get("recommendation") or "", } ) return out def summarize(findings: list[dict]) -> dict: conf_crit = sum( 1 for f in findings if f["status"] == "confirmed" and f["severity"] == "critical" ) conf_high = sum( 1 for f in findings if f["status"] == "confirmed" and f["severity"] == "high" ) return { "confirmed_critical": conf_crit, "confirmed_high": conf_high, "block": (conf_crit + conf_high) > 0, } async def main_async(args: argparse.Namespace) -> int: target = Path(args.target).resolve() if not target.is_dir(): log(f"target dir not found: {target}") return 2 scope = args.scope.strip() if args.scope else "the entire repository" selected = ( [d.strip() for d in args.detectors.split(",") if d.strip()] if args.detectors else list(DETECTORS) ) unknown = [d for d in selected if d not in DETECTORS] if unknown: log(f"unknown detector(s): {unknown}; valid: {list(DETECTORS)}") return 2 budget = Budget(args.total_budget_usd) sem = asyncio.Semaphore(max(1, args.concurrency)) log( f"target={target} scope='{scope}' detectors={selected} model={args.model or 'cli-default'} " f"concurrency={args.concurrency} total_budget=${args.total_budget_usd}" ) det_results = await asyncio.gather( *[ run_detector( cat, scope, cwd=target, model=args.model, max_turns=args.max_turns, budget_usd=args.detector_budget_usd, sem=sem, budget=budget, ) for cat in selected ] ) candidates: list[dict] = [] errors: list[str] = [] for cat, found, err in det_results: candidates.extend(found) if err: errors.append(f"{cat}: {err}") log( f"total candidates: {len(candidates)}; detector spend so far: ${budget.spent:.3f}" ) if candidates and not budget.exhausted(): verdicts, _, verr = await run_verifier( candidates, cwd=target, model=args.model, max_turns=args.max_turns, budget_usd=args.detector_budget_usd, budget=budget, ) if verr: errors.append(f"verifier: {verr}") else: verdicts = {} if budget.exhausted(): errors.append( "verifier: skipped (budget exhausted) — all candidates left unverified" ) findings = merge(candidates, verdicts) report = { "findings": findings, "summary": summarize(findings), "_meta": { "target": str(target), "scope": scope, "detectors": selected, "model": args.model or "cli-default", "spend_usd": round(budget.spent, 4), "errors": errors, }, } out = json.dumps(report, indent=2) if args.out: Path(args.out).write_text(out) log(f"wrote {args.out}") else: print(out) s = report["summary"] log( f"DONE: {s['confirmed_critical']} confirmed-crit, {s['confirmed_high']} confirmed-high, " f"block={s['block']}, spend=${budget.spent:.3f}, errors={len(errors)}" ) if errors: for e in errors: log(f" ERROR/NOTE: {e}") return 0 def main() -> None: p = argparse.ArgumentParser( description="Headless Sea Haven security detector fan-out + verifier" ) p.add_argument("target", help="target repository directory") p.add_argument( "--scope", default="", help="space-separated subdirs to restrict the audit" ) p.add_argument( "--out", default="", help="write findings JSON here (default: stdout)" ) p.add_argument("--model", default=None, help="model id (default: CLI default)") p.add_argument( "--detectors", default="", help="comma list to restrict detectors (default: all 6)", ) p.add_argument( "--concurrency", type=int, default=3, help="max concurrent detectors" ) p.add_argument( "--max-turns", type=int, default=40, help="max agent turns per detector/verifier", ) p.add_argument( "--detector-budget-usd", type=float, default=2.0, help="per-call SDK spend cap" ) p.add_argument( "--total-budget-usd", type=float, default=12.0, help="overall spend cap (0 = unlimited)", ) args = p.parse_args() # Guarantee the subscription OAuth path: a raw API key would silently win, so remove it. if os.environ.pop("ANTHROPIC_API_KEY", None): log("removed ANTHROPIC_API_KEY from env to force the subscription OAuth path") if not os.environ.get("CLAUDE_CODE_OAUTH_TOKEN"): log( "FATAL: CLAUDE_CODE_OAUTH_TOKEN not set (source ~/secrev.env). Refusing to run." ) sys.exit(2) sys.exit(asyncio.run(main_async(args))) if __name__ == "__main__": main()