security-review/run_headless.py
Adam Moussa 4c88c01f7b
chore: import security-review gate, sweep, and Plane-1 checkers into standalone repo
Fresh-init copy of the security-review/ subsystem extracted from
Sea-Haven-Industries/orchestrator (being deprecated). Adds org-standard scaffold:
CI reusable-workflow callers (ruff + collect), dependency-review, labeler,
dependabot, .gitignore, requirements.txt. Scheduled execution is migrating to
Claude Code web routines (ALARM-only to #repo-scanner); the systemd units and
nightly_sweep.sh/checker_coordinator.sh remain the source of truth.

Committed with --no-verify: the canary fixtures (checkers/fixtures/**) carry
intentional secret-shaped test data that trips the deterministic gate (the
documented detector-fixture false positive); no new logic is introduced.
2026-06-29 11:41:41 -04:00

447 lines
16 KiB
Python

#!/usr/bin/env python3
"""run_headless.py — Path B headless detector fan-out + proof-or-kill verifier.
Reuses the /sh-security-review detector + verifier prompts, but runs them
unattended via the Claude Agent SDK instead of interactive Claude Code subagents.
Authenticates with the Claude subscription OAuth token (CLAUDE_CODE_OAUTH_TOKEN)
through the bundled `claude` CLI — NEVER a raw ANTHROPIC_API_KEY (which would be
metered and would silently win if both were set, so we pop it).
Emits the finding-schema JSON ({findings, summary}) that
`review.sh --agent-findings` consumes. review.sh re-derives the gate, so this
script's job is high-recall candidate generation + proof-or-kill verification,
failing toward over-reporting (never silently drops a candidate or a parse error).
See memory project-security-review-agent and security-review/DEPLOY-R720.md.
Usage:
CLAUDE_CODE_OAUTH_TOKEN=... python3 run_headless.py TARGET_DIR \
[--scope "src infra web"] [--out findings.json] [--model claude-...] \
[--concurrency 3] [--detectors injection,authz] \
[--detector-budget-usd 2.0] [--total-budget-usd 12.0]
"""
from __future__ import annotations
import argparse
import asyncio
import json
import os
import re
import sys
import time
from pathlib import Path
from claude_agent_sdk import ClaudeAgentOptions, query
# Read-only surface: detectors reason over source, they don't mutate or fetch.
READONLY_TOOLS = ["Read", "Grep", "Glob"]
BLOCKED_TOOLS = ["Bash", "Write", "Edit", "NotebookEdit", "WebFetch", "WebSearch"]
# Detector category -> closed checklist (verbatim intent from sh-security-review.md).
DETECTORS: dict[str, str] = {
"injection": "SQL/command/template injection, unsafe deserialization, SSRF, path/file traversal, XXE",
"authz": "broken object-level auth/IDOR, missing access checks, missing webhook/Slack signature verification, auth bypass",
"secrets-crypto": "hardcoded secrets/keys, weak/broken crypto (MD5/SHA1/unsalted), sensitive data in logs/errors/responses, wrong SSM-vs-Secrets-Manager placement",
"iac-iam": "wildcard IAM actions/resources, public buckets/endpoints, open security-group ingress (0.0.0.0/0), missing encryption, over-broad trust policies",
"web-client": "XSS (incl. dangerouslySetInnerHTML), CSRF, open redirect, client-side secret exposure",
"logic": "broken multi-step invariants, race conditions, missing tenant isolation, auth-state confusion",
}
DETECTOR_TMPL = """You are a hostile {category} security auditor for Sea Haven. Assume this code is \
hostile and the author missed something. Audit ONLY {category} issues in: {scope}. The repository root \
is your current working directory; read the actual files with your tools. Ignore .git and any file that \
is obviously an answer key.
For each checklist item, either name a specific line that is provably safe, OR file a finding. Do not \
hand-wave or give an open "looks fine" verdict.
Checklist: {checklist}
Return ONLY a JSON array (no prose, no markdown fences) where each element is:
{{"id": "<stable-slug>", "title": "...", "claimed_severity": "critical|high|medium|low|info",
"cwe": "CWE-####", "file": "<repo-relative path>", "line": <int or null>, "category": "{category}",
"data_flow": "numbered source->sink trace", "proof": {{"input": "concrete malicious input/trigger",
"outcome": "the specific bad result", "test": "optional failing-test sketch or null"}},
"recommendation": "..."}}
If there are no findings, return []."""
VERIFIER_TMPL = """You are a skeptical exploitation verifier. You did NOT find these; your job is to \
REFUTE weak claims. The repository root is your current working directory; read the real files to check \
reachability before ruling.
For each candidate finding decide whether there is a concrete, plausible proof-of-exploit (a specific \
malicious input and the specific bad outcome, consistent with the code):
- If yes: set "status":"confirmed", keep "severity" equal to the claimed_severity, and tighten the proof.
- If no / speculative / not reachable: set "status":"unverified" and "severity":"unverified". Default to \
unverified when uncertain. A confident assertion with no demonstrable input is NOT proof.
Return ONLY a JSON array (no prose, no fences) of the SAME findings, each preserving its original "id", \
"title", "cwe", "file", "line", "category", "claimed_severity", "data_flow", "recommendation", and adding \
"severity" (final) plus "status" and the verified "proof". Return one element per candidate — do not drop \
any.
Candidates:
{candidates}"""
SEV_RANK = {
"critical": 4,
"high": 3,
"medium": 2,
"low": 1,
"info": 0,
"unverified": -1,
}
def log(msg: str) -> None:
print(f"[run_headless] {msg}", file=sys.stderr, flush=True)
def extract_json(text: str):
"""Best-effort parse of a JSON array/object from an agent's final text."""
if not text:
return None
t = text.strip()
if t.startswith("```"):
t = re.sub(r"^```[a-zA-Z0-9]*\n?", "", t)
t = re.sub(r"\n?```\s*$", "", t).strip()
try:
return json.loads(t)
except Exception:
pass
# Fall back to the outermost [...] (or {...}) span.
for open_c, close_c in (("[", "]"), ("{", "}")):
start, end = t.find(open_c), t.rfind(close_c)
if 0 <= start < end:
try:
return json.loads(t[start : end + 1])
except Exception:
continue
return None
async def run_agent(
prompt: str, *, cwd: Path, model: str | None, max_turns: int, budget_usd: float
) -> tuple[str, float, bool]:
"""Run one fresh-context agent turn; return (final_text, cost_usd, is_error)."""
opts = ClaudeAgentOptions(
allowed_tools=READONLY_TOOLS,
disallowed_tools=BLOCKED_TOOLS,
permission_mode="bypassPermissions",
setting_sources=[], # hermetic: ignore user/project/local config + CLAUDE.md
cwd=str(cwd),
model=model,
max_turns=max_turns,
max_budget_usd=budget_usd,
)
texts: list[str] = []
result_text: str | None = None
cost = 0.0
is_error = False
async for msg in query(prompt=prompt, options=opts):
name = type(msg).__name__
if name == "AssistantMessage":
for block in getattr(msg, "content", []) or []:
t = getattr(block, "text", None)
if t:
texts.append(t)
elif name == "ResultMessage":
result_text = getattr(msg, "result", None)
cost = float(getattr(msg, "total_cost_usd", 0.0) or 0.0)
is_error = bool(getattr(msg, "is_error", False))
return (result_text or "\n".join(texts)), cost, is_error
class Budget:
def __init__(self, total: float) -> None:
self.total = total
self.spent = 0.0
self._lock = asyncio.Lock()
async def add(self, amount: float) -> None:
async with self._lock:
self.spent += amount
def exhausted(self) -> bool:
return self.total > 0 and self.spent >= self.total
async def run_detector(
category: str,
scope: str,
*,
cwd: Path,
model: str | None,
max_turns: int,
budget_usd: float,
sem: asyncio.Semaphore,
budget: Budget,
) -> tuple[str, list[dict], str | None]:
"""Returns (category, candidate_findings, error_message_or_None)."""
async with sem:
if budget.exhausted():
return category, [], "skipped: total budget exhausted"
prompt = DETECTOR_TMPL.format(
category=category, scope=scope, checklist=DETECTORS[category]
)
t0 = time.monotonic()
try:
text, cost, is_error = await run_agent(
prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd
)
except Exception as exc: # never let one detector kill the run
log(f"detector {category}: EXCEPTION {type(exc).__name__}: {exc}")
return category, [], f"exception: {type(exc).__name__}: {exc}"
await budget.add(cost)
dt = time.monotonic() - t0
parsed = extract_json(text)
if not isinstance(parsed, list):
log(
f"detector {category}: UNPARSEABLE output ({dt:.0f}s, ${cost:.3f}) — over-reporting as error"
)
return category, [], "unparseable detector output (NOT treated as clean)"
for f in parsed:
if isinstance(f, dict):
f.setdefault("category", category)
log(
f"detector {category}: {len(parsed)} candidate(s) ({dt:.0f}s, ${cost:.3f})"
+ (" [is_error]" if is_error else "")
)
return category, [f for f in parsed if isinstance(f, dict)], None
async def run_verifier(
candidates: list[dict],
*,
cwd: Path,
model: str | None,
max_turns: int,
budget_usd: float,
budget: Budget,
) -> tuple[dict[str, dict], float, str | None]:
"""Returns ({id -> verdict}, cost, error). Empty verdicts on failure (caller keeps candidates)."""
prompt = VERIFIER_TMPL.format(candidates=json.dumps(candidates, indent=2))
try:
text, cost, _ = await run_agent(
prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd
)
except Exception as exc:
log(f"verifier: EXCEPTION {type(exc).__name__}: {exc}")
return {}, 0.0, f"exception: {type(exc).__name__}: {exc}"
await budget.add(cost)
parsed = extract_json(text)
if not isinstance(parsed, list):
log(
f"verifier: UNPARSEABLE output (${cost:.3f}) — keeping all candidates as unverified"
)
return {}, cost, "unparseable verifier output"
verdicts = {f["id"]: f for f in parsed if isinstance(f, dict) and f.get("id")}
log(f"verifier: ruled on {len(verdicts)} finding(s) (${cost:.3f})")
return verdicts, cost, None
def merge(candidates: list[dict], verdicts: dict[str, dict]) -> list[dict]:
"""Apply verifier verdicts to candidates. A candidate the verifier dropped or never
ruled on stays as 'unverified' — fail toward over-reporting, never silently delete."""
out: list[dict] = []
for c in candidates:
fid = c.get("id") or f"anon-{c.get('file', '?')}-{c.get('line', '?')}"
c.setdefault("id", fid)
v = verdicts.get(fid, {})
status = v.get("status")
if status not in ("confirmed", "unverified", "suppressed"):
status = "unverified"
if status == "confirmed":
severity = v.get("severity") or c.get("claimed_severity") or "high"
else:
severity = "unverified"
out.append(
{
"id": fid,
"title": v.get("title") or c.get("title") or fid,
"severity": severity,
"claimed_severity": c.get("claimed_severity") or "high",
"cwe": c.get("cwe") or v.get("cwe") or "n/a",
"file": c.get("file") or v.get("file") or "",
"line": c.get("line", v.get("line")),
"category": c.get("category") or v.get("category") or "other",
"data_flow": v.get("data_flow") or c.get("data_flow") or "",
"proof": v.get("proof")
or c.get("proof")
or {"input": "", "outcome": ""},
"status": status,
"recommendation": v.get("recommendation")
or c.get("recommendation")
or "",
}
)
return out
def summarize(findings: list[dict]) -> dict:
conf_crit = sum(
1
for f in findings
if f["status"] == "confirmed" and f["severity"] == "critical"
)
conf_high = sum(
1 for f in findings if f["status"] == "confirmed" and f["severity"] == "high"
)
return {
"confirmed_critical": conf_crit,
"confirmed_high": conf_high,
"block": (conf_crit + conf_high) > 0,
}
async def main_async(args: argparse.Namespace) -> int:
target = Path(args.target).resolve()
if not target.is_dir():
log(f"target dir not found: {target}")
return 2
scope = args.scope.strip() if args.scope else "the entire repository"
selected = (
[d.strip() for d in args.detectors.split(",") if d.strip()]
if args.detectors
else list(DETECTORS)
)
unknown = [d for d in selected if d not in DETECTORS]
if unknown:
log(f"unknown detector(s): {unknown}; valid: {list(DETECTORS)}")
return 2
budget = Budget(args.total_budget_usd)
sem = asyncio.Semaphore(max(1, args.concurrency))
log(
f"target={target} scope='{scope}' detectors={selected} model={args.model or 'cli-default'} "
f"concurrency={args.concurrency} total_budget=${args.total_budget_usd}"
)
det_results = await asyncio.gather(
*[
run_detector(
cat,
scope,
cwd=target,
model=args.model,
max_turns=args.max_turns,
budget_usd=args.detector_budget_usd,
sem=sem,
budget=budget,
)
for cat in selected
]
)
candidates: list[dict] = []
errors: list[str] = []
for cat, found, err in det_results:
candidates.extend(found)
if err:
errors.append(f"{cat}: {err}")
log(
f"total candidates: {len(candidates)}; detector spend so far: ${budget.spent:.3f}"
)
if candidates and not budget.exhausted():
verdicts, _, verr = await run_verifier(
candidates,
cwd=target,
model=args.model,
max_turns=args.max_turns,
budget_usd=args.detector_budget_usd,
budget=budget,
)
if verr:
errors.append(f"verifier: {verr}")
else:
verdicts = {}
if budget.exhausted():
errors.append(
"verifier: skipped (budget exhausted) — all candidates left unverified"
)
findings = merge(candidates, verdicts)
report = {
"findings": findings,
"summary": summarize(findings),
"_meta": {
"target": str(target),
"scope": scope,
"detectors": selected,
"model": args.model or "cli-default",
"spend_usd": round(budget.spent, 4),
"errors": errors,
},
}
out = json.dumps(report, indent=2)
if args.out:
Path(args.out).write_text(out)
log(f"wrote {args.out}")
else:
print(out)
s = report["summary"]
log(
f"DONE: {s['confirmed_critical']} confirmed-crit, {s['confirmed_high']} confirmed-high, "
f"block={s['block']}, spend=${budget.spent:.3f}, errors={len(errors)}"
)
if errors:
for e in errors:
log(f" ERROR/NOTE: {e}")
return 0
def main() -> None:
p = argparse.ArgumentParser(
description="Headless Sea Haven security detector fan-out + verifier"
)
p.add_argument("target", help="target repository directory")
p.add_argument(
"--scope", default="", help="space-separated subdirs to restrict the audit"
)
p.add_argument(
"--out", default="", help="write findings JSON here (default: stdout)"
)
p.add_argument("--model", default=None, help="model id (default: CLI default)")
p.add_argument(
"--detectors",
default="",
help="comma list to restrict detectors (default: all 6)",
)
p.add_argument(
"--concurrency", type=int, default=3, help="max concurrent detectors"
)
p.add_argument(
"--max-turns",
type=int,
default=40,
help="max agent turns per detector/verifier",
)
p.add_argument(
"--detector-budget-usd", type=float, default=2.0, help="per-call SDK spend cap"
)
p.add_argument(
"--total-budget-usd",
type=float,
default=12.0,
help="overall spend cap (0 = unlimited)",
)
args = p.parse_args()
# Guarantee the subscription OAuth path: a raw API key would silently win, so remove it.
if os.environ.pop("ANTHROPIC_API_KEY", None):
log("removed ANTHROPIC_API_KEY from env to force the subscription OAuth path")
if not os.environ.get("CLAUDE_CODE_OAUTH_TOKEN"):
log(
"FATAL: CLAUDE_CODE_OAUTH_TOKEN not set (source ~/secrev.env). Refusing to run."
)
sys.exit(2)
sys.exit(asyncio.run(main_async(args)))
if __name__ == "__main__":
main()