Add headless detector fan-out + proof-or-kill verifier runner
run_headless.py is the Path B engine the nightly sweep calls: the 6 fresh-context detectors + proof-or-kill verifier from /sh-security-review, run unattended via the Claude Agent SDK on subscription OAuth (pops ANTHROPIC_API_KEY so the API key can't silently win). Read-only tools, hermetic, per-call + total budget caps, fails toward over-reporting. Emits the finding schema that review.sh --agent-findings consumes.
This commit is contained in:
parent
a05afb5ebc
commit
f4dd72ced9
1 changed files with 447 additions and 0 deletions
447
security-review/run_headless.py
Normal file
447
security-review/run_headless.py
Normal file
|
|
@ -0,0 +1,447 @@
|
|||
#!/usr/bin/env python3
|
||||
"""run_headless.py — Path B headless detector fan-out + proof-or-kill verifier.
|
||||
|
||||
Reuses the /sh-security-review detector + verifier prompts, but runs them
|
||||
unattended via the Claude Agent SDK instead of interactive Claude Code subagents.
|
||||
Authenticates with the Claude subscription OAuth token (CLAUDE_CODE_OAUTH_TOKEN)
|
||||
through the bundled `claude` CLI — NEVER a raw ANTHROPIC_API_KEY (which would be
|
||||
metered and would silently win if both were set, so we pop it).
|
||||
|
||||
Emits the finding-schema JSON ({findings, summary}) that
|
||||
`review.sh --agent-findings` consumes. review.sh re-derives the gate, so this
|
||||
script's job is high-recall candidate generation + proof-or-kill verification,
|
||||
failing toward over-reporting (never silently drops a candidate or a parse error).
|
||||
|
||||
See memory project-security-review-agent and security-review/DEPLOY-R720.md.
|
||||
|
||||
Usage:
|
||||
CLAUDE_CODE_OAUTH_TOKEN=... python3 run_headless.py TARGET_DIR \
|
||||
[--scope "src infra web"] [--out findings.json] [--model claude-...] \
|
||||
[--concurrency 3] [--detectors injection,authz] \
|
||||
[--detector-budget-usd 2.0] [--total-budget-usd 12.0]
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
from claude_agent_sdk import ClaudeAgentOptions, query
|
||||
|
||||
# Read-only surface: detectors reason over source, they don't mutate or fetch.
|
||||
READONLY_TOOLS = ["Read", "Grep", "Glob"]
|
||||
BLOCKED_TOOLS = ["Bash", "Write", "Edit", "NotebookEdit", "WebFetch", "WebSearch"]
|
||||
|
||||
# Detector category -> closed checklist (verbatim intent from sh-security-review.md).
|
||||
DETECTORS: dict[str, str] = {
|
||||
"injection": "SQL/command/template injection, unsafe deserialization, SSRF, path/file traversal, XXE",
|
||||
"authz": "broken object-level auth/IDOR, missing access checks, missing webhook/Slack signature verification, auth bypass",
|
||||
"secrets-crypto": "hardcoded secrets/keys, weak/broken crypto (MD5/SHA1/unsalted), sensitive data in logs/errors/responses, wrong SSM-vs-Secrets-Manager placement",
|
||||
"iac-iam": "wildcard IAM actions/resources, public buckets/endpoints, open security-group ingress (0.0.0.0/0), missing encryption, over-broad trust policies",
|
||||
"web-client": "XSS (incl. dangerouslySetInnerHTML), CSRF, open redirect, client-side secret exposure",
|
||||
"logic": "broken multi-step invariants, race conditions, missing tenant isolation, auth-state confusion",
|
||||
}
|
||||
|
||||
DETECTOR_TMPL = """You are a hostile {category} security auditor for Sea Haven. Assume this code is \
|
||||
hostile and the author missed something. Audit ONLY {category} issues in: {scope}. The repository root \
|
||||
is your current working directory; read the actual files with your tools. Ignore .git and any file that \
|
||||
is obviously an answer key.
|
||||
|
||||
For each checklist item, either name a specific line that is provably safe, OR file a finding. Do not \
|
||||
hand-wave or give an open "looks fine" verdict.
|
||||
|
||||
Checklist: {checklist}
|
||||
|
||||
Return ONLY a JSON array (no prose, no markdown fences) where each element is:
|
||||
{{"id": "<stable-slug>", "title": "...", "claimed_severity": "critical|high|medium|low|info",
|
||||
"cwe": "CWE-####", "file": "<repo-relative path>", "line": <int or null>, "category": "{category}",
|
||||
"data_flow": "numbered source->sink trace", "proof": {{"input": "concrete malicious input/trigger",
|
||||
"outcome": "the specific bad result", "test": "optional failing-test sketch or null"}},
|
||||
"recommendation": "..."}}
|
||||
If there are no findings, return []."""
|
||||
|
||||
VERIFIER_TMPL = """You are a skeptical exploitation verifier. You did NOT find these; your job is to \
|
||||
REFUTE weak claims. The repository root is your current working directory; read the real files to check \
|
||||
reachability before ruling.
|
||||
|
||||
For each candidate finding decide whether there is a concrete, plausible proof-of-exploit (a specific \
|
||||
malicious input and the specific bad outcome, consistent with the code):
|
||||
- If yes: set "status":"confirmed", keep "severity" equal to the claimed_severity, and tighten the proof.
|
||||
- If no / speculative / not reachable: set "status":"unverified" and "severity":"unverified". Default to \
|
||||
unverified when uncertain. A confident assertion with no demonstrable input is NOT proof.
|
||||
|
||||
Return ONLY a JSON array (no prose, no fences) of the SAME findings, each preserving its original "id", \
|
||||
"title", "cwe", "file", "line", "category", "claimed_severity", "data_flow", "recommendation", and adding \
|
||||
"severity" (final) plus "status" and the verified "proof". Return one element per candidate — do not drop \
|
||||
any.
|
||||
|
||||
Candidates:
|
||||
{candidates}"""
|
||||
|
||||
SEV_RANK = {
|
||||
"critical": 4,
|
||||
"high": 3,
|
||||
"medium": 2,
|
||||
"low": 1,
|
||||
"info": 0,
|
||||
"unverified": -1,
|
||||
}
|
||||
|
||||
|
||||
def log(msg: str) -> None:
|
||||
print(f"[run_headless] {msg}", file=sys.stderr, flush=True)
|
||||
|
||||
|
||||
def extract_json(text: str):
|
||||
"""Best-effort parse of a JSON array/object from an agent's final text."""
|
||||
if not text:
|
||||
return None
|
||||
t = text.strip()
|
||||
if t.startswith("```"):
|
||||
t = re.sub(r"^```[a-zA-Z0-9]*\n?", "", t)
|
||||
t = re.sub(r"\n?```\s*$", "", t).strip()
|
||||
try:
|
||||
return json.loads(t)
|
||||
except Exception:
|
||||
pass
|
||||
# Fall back to the outermost [...] (or {...}) span.
|
||||
for open_c, close_c in (("[", "]"), ("{", "}")):
|
||||
start, end = t.find(open_c), t.rfind(close_c)
|
||||
if 0 <= start < end:
|
||||
try:
|
||||
return json.loads(t[start : end + 1])
|
||||
except Exception:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
async def run_agent(
|
||||
prompt: str, *, cwd: Path, model: str | None, max_turns: int, budget_usd: float
|
||||
) -> tuple[str, float, bool]:
|
||||
"""Run one fresh-context agent turn; return (final_text, cost_usd, is_error)."""
|
||||
opts = ClaudeAgentOptions(
|
||||
allowed_tools=READONLY_TOOLS,
|
||||
disallowed_tools=BLOCKED_TOOLS,
|
||||
permission_mode="bypassPermissions",
|
||||
setting_sources=[], # hermetic: ignore user/project/local config + CLAUDE.md
|
||||
cwd=str(cwd),
|
||||
model=model,
|
||||
max_turns=max_turns,
|
||||
max_budget_usd=budget_usd,
|
||||
)
|
||||
texts: list[str] = []
|
||||
result_text: str | None = None
|
||||
cost = 0.0
|
||||
is_error = False
|
||||
async for msg in query(prompt=prompt, options=opts):
|
||||
name = type(msg).__name__
|
||||
if name == "AssistantMessage":
|
||||
for block in getattr(msg, "content", []) or []:
|
||||
t = getattr(block, "text", None)
|
||||
if t:
|
||||
texts.append(t)
|
||||
elif name == "ResultMessage":
|
||||
result_text = getattr(msg, "result", None)
|
||||
cost = float(getattr(msg, "total_cost_usd", 0.0) or 0.0)
|
||||
is_error = bool(getattr(msg, "is_error", False))
|
||||
return (result_text or "\n".join(texts)), cost, is_error
|
||||
|
||||
|
||||
class Budget:
|
||||
def __init__(self, total: float) -> None:
|
||||
self.total = total
|
||||
self.spent = 0.0
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
async def add(self, amount: float) -> None:
|
||||
async with self._lock:
|
||||
self.spent += amount
|
||||
|
||||
def exhausted(self) -> bool:
|
||||
return self.total > 0 and self.spent >= self.total
|
||||
|
||||
|
||||
async def run_detector(
|
||||
category: str,
|
||||
scope: str,
|
||||
*,
|
||||
cwd: Path,
|
||||
model: str | None,
|
||||
max_turns: int,
|
||||
budget_usd: float,
|
||||
sem: asyncio.Semaphore,
|
||||
budget: Budget,
|
||||
) -> tuple[str, list[dict], str | None]:
|
||||
"""Returns (category, candidate_findings, error_message_or_None)."""
|
||||
async with sem:
|
||||
if budget.exhausted():
|
||||
return category, [], "skipped: total budget exhausted"
|
||||
prompt = DETECTOR_TMPL.format(
|
||||
category=category, scope=scope, checklist=DETECTORS[category]
|
||||
)
|
||||
t0 = time.monotonic()
|
||||
try:
|
||||
text, cost, is_error = await run_agent(
|
||||
prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd
|
||||
)
|
||||
except Exception as exc: # never let one detector kill the run
|
||||
log(f"detector {category}: EXCEPTION {type(exc).__name__}: {exc}")
|
||||
return category, [], f"exception: {type(exc).__name__}: {exc}"
|
||||
await budget.add(cost)
|
||||
dt = time.monotonic() - t0
|
||||
parsed = extract_json(text)
|
||||
if not isinstance(parsed, list):
|
||||
log(
|
||||
f"detector {category}: UNPARSEABLE output ({dt:.0f}s, ${cost:.3f}) — over-reporting as error"
|
||||
)
|
||||
return category, [], "unparseable detector output (NOT treated as clean)"
|
||||
for f in parsed:
|
||||
if isinstance(f, dict):
|
||||
f.setdefault("category", category)
|
||||
log(
|
||||
f"detector {category}: {len(parsed)} candidate(s) ({dt:.0f}s, ${cost:.3f})"
|
||||
+ (" [is_error]" if is_error else "")
|
||||
)
|
||||
return category, [f for f in parsed if isinstance(f, dict)], None
|
||||
|
||||
|
||||
async def run_verifier(
|
||||
candidates: list[dict],
|
||||
*,
|
||||
cwd: Path,
|
||||
model: str | None,
|
||||
max_turns: int,
|
||||
budget_usd: float,
|
||||
budget: Budget,
|
||||
) -> tuple[dict[str, dict], float, str | None]:
|
||||
"""Returns ({id -> verdict}, cost, error). Empty verdicts on failure (caller keeps candidates)."""
|
||||
prompt = VERIFIER_TMPL.format(candidates=json.dumps(candidates, indent=2))
|
||||
try:
|
||||
text, cost, _ = await run_agent(
|
||||
prompt, cwd=cwd, model=model, max_turns=max_turns, budget_usd=budget_usd
|
||||
)
|
||||
except Exception as exc:
|
||||
log(f"verifier: EXCEPTION {type(exc).__name__}: {exc}")
|
||||
return {}, 0.0, f"exception: {type(exc).__name__}: {exc}"
|
||||
await budget.add(cost)
|
||||
parsed = extract_json(text)
|
||||
if not isinstance(parsed, list):
|
||||
log(
|
||||
f"verifier: UNPARSEABLE output (${cost:.3f}) — keeping all candidates as unverified"
|
||||
)
|
||||
return {}, cost, "unparseable verifier output"
|
||||
verdicts = {f["id"]: f for f in parsed if isinstance(f, dict) and f.get("id")}
|
||||
log(f"verifier: ruled on {len(verdicts)} finding(s) (${cost:.3f})")
|
||||
return verdicts, cost, None
|
||||
|
||||
|
||||
def merge(candidates: list[dict], verdicts: dict[str, dict]) -> list[dict]:
|
||||
"""Apply verifier verdicts to candidates. A candidate the verifier dropped or never
|
||||
ruled on stays as 'unverified' — fail toward over-reporting, never silently delete."""
|
||||
out: list[dict] = []
|
||||
for c in candidates:
|
||||
fid = c.get("id") or f"anon-{c.get('file', '?')}-{c.get('line', '?')}"
|
||||
c.setdefault("id", fid)
|
||||
v = verdicts.get(fid, {})
|
||||
status = v.get("status")
|
||||
if status not in ("confirmed", "unverified", "suppressed"):
|
||||
status = "unverified"
|
||||
if status == "confirmed":
|
||||
severity = v.get("severity") or c.get("claimed_severity") or "high"
|
||||
else:
|
||||
severity = "unverified"
|
||||
out.append(
|
||||
{
|
||||
"id": fid,
|
||||
"title": v.get("title") or c.get("title") or fid,
|
||||
"severity": severity,
|
||||
"claimed_severity": c.get("claimed_severity") or "high",
|
||||
"cwe": c.get("cwe") or v.get("cwe") or "n/a",
|
||||
"file": c.get("file") or v.get("file") or "",
|
||||
"line": c.get("line", v.get("line")),
|
||||
"category": c.get("category") or v.get("category") or "other",
|
||||
"data_flow": v.get("data_flow") or c.get("data_flow") or "",
|
||||
"proof": v.get("proof")
|
||||
or c.get("proof")
|
||||
or {"input": "", "outcome": ""},
|
||||
"status": status,
|
||||
"recommendation": v.get("recommendation")
|
||||
or c.get("recommendation")
|
||||
or "",
|
||||
}
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
def summarize(findings: list[dict]) -> dict:
|
||||
conf_crit = sum(
|
||||
1
|
||||
for f in findings
|
||||
if f["status"] == "confirmed" and f["severity"] == "critical"
|
||||
)
|
||||
conf_high = sum(
|
||||
1 for f in findings if f["status"] == "confirmed" and f["severity"] == "high"
|
||||
)
|
||||
return {
|
||||
"confirmed_critical": conf_crit,
|
||||
"confirmed_high": conf_high,
|
||||
"block": (conf_crit + conf_high) > 0,
|
||||
}
|
||||
|
||||
|
||||
async def main_async(args: argparse.Namespace) -> int:
|
||||
target = Path(args.target).resolve()
|
||||
if not target.is_dir():
|
||||
log(f"target dir not found: {target}")
|
||||
return 2
|
||||
scope = args.scope.strip() if args.scope else "the entire repository"
|
||||
selected = (
|
||||
[d.strip() for d in args.detectors.split(",") if d.strip()]
|
||||
if args.detectors
|
||||
else list(DETECTORS)
|
||||
)
|
||||
unknown = [d for d in selected if d not in DETECTORS]
|
||||
if unknown:
|
||||
log(f"unknown detector(s): {unknown}; valid: {list(DETECTORS)}")
|
||||
return 2
|
||||
|
||||
budget = Budget(args.total_budget_usd)
|
||||
sem = asyncio.Semaphore(max(1, args.concurrency))
|
||||
log(
|
||||
f"target={target} scope='{scope}' detectors={selected} model={args.model or 'cli-default'} "
|
||||
f"concurrency={args.concurrency} total_budget=${args.total_budget_usd}"
|
||||
)
|
||||
|
||||
det_results = await asyncio.gather(
|
||||
*[
|
||||
run_detector(
|
||||
cat,
|
||||
scope,
|
||||
cwd=target,
|
||||
model=args.model,
|
||||
max_turns=args.max_turns,
|
||||
budget_usd=args.detector_budget_usd,
|
||||
sem=sem,
|
||||
budget=budget,
|
||||
)
|
||||
for cat in selected
|
||||
]
|
||||
)
|
||||
|
||||
candidates: list[dict] = []
|
||||
errors: list[str] = []
|
||||
for cat, found, err in det_results:
|
||||
candidates.extend(found)
|
||||
if err:
|
||||
errors.append(f"{cat}: {err}")
|
||||
|
||||
log(
|
||||
f"total candidates: {len(candidates)}; detector spend so far: ${budget.spent:.3f}"
|
||||
)
|
||||
|
||||
if candidates and not budget.exhausted():
|
||||
verdicts, _, verr = await run_verifier(
|
||||
candidates,
|
||||
cwd=target,
|
||||
model=args.model,
|
||||
max_turns=args.max_turns,
|
||||
budget_usd=args.detector_budget_usd,
|
||||
budget=budget,
|
||||
)
|
||||
if verr:
|
||||
errors.append(f"verifier: {verr}")
|
||||
else:
|
||||
verdicts = {}
|
||||
if budget.exhausted():
|
||||
errors.append(
|
||||
"verifier: skipped (budget exhausted) — all candidates left unverified"
|
||||
)
|
||||
|
||||
findings = merge(candidates, verdicts)
|
||||
report = {
|
||||
"findings": findings,
|
||||
"summary": summarize(findings),
|
||||
"_meta": {
|
||||
"target": str(target),
|
||||
"scope": scope,
|
||||
"detectors": selected,
|
||||
"model": args.model or "cli-default",
|
||||
"spend_usd": round(budget.spent, 4),
|
||||
"errors": errors,
|
||||
},
|
||||
}
|
||||
out = json.dumps(report, indent=2)
|
||||
if args.out:
|
||||
Path(args.out).write_text(out)
|
||||
log(f"wrote {args.out}")
|
||||
else:
|
||||
print(out)
|
||||
|
||||
s = report["summary"]
|
||||
log(
|
||||
f"DONE: {s['confirmed_critical']} confirmed-crit, {s['confirmed_high']} confirmed-high, "
|
||||
f"block={s['block']}, spend=${budget.spent:.3f}, errors={len(errors)}"
|
||||
)
|
||||
if errors:
|
||||
for e in errors:
|
||||
log(f" ERROR/NOTE: {e}")
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> None:
|
||||
p = argparse.ArgumentParser(
|
||||
description="Headless Sea Haven security detector fan-out + verifier"
|
||||
)
|
||||
p.add_argument("target", help="target repository directory")
|
||||
p.add_argument(
|
||||
"--scope", default="", help="space-separated subdirs to restrict the audit"
|
||||
)
|
||||
p.add_argument(
|
||||
"--out", default="", help="write findings JSON here (default: stdout)"
|
||||
)
|
||||
p.add_argument("--model", default=None, help="model id (default: CLI default)")
|
||||
p.add_argument(
|
||||
"--detectors",
|
||||
default="",
|
||||
help="comma list to restrict detectors (default: all 6)",
|
||||
)
|
||||
p.add_argument(
|
||||
"--concurrency", type=int, default=3, help="max concurrent detectors"
|
||||
)
|
||||
p.add_argument(
|
||||
"--max-turns",
|
||||
type=int,
|
||||
default=40,
|
||||
help="max agent turns per detector/verifier",
|
||||
)
|
||||
p.add_argument(
|
||||
"--detector-budget-usd", type=float, default=2.0, help="per-call SDK spend cap"
|
||||
)
|
||||
p.add_argument(
|
||||
"--total-budget-usd",
|
||||
type=float,
|
||||
default=12.0,
|
||||
help="overall spend cap (0 = unlimited)",
|
||||
)
|
||||
args = p.parse_args()
|
||||
|
||||
# Guarantee the subscription OAuth path: a raw API key would silently win, so remove it.
|
||||
if os.environ.pop("ANTHROPIC_API_KEY", None):
|
||||
log("removed ANTHROPIC_API_KEY from env to force the subscription OAuth path")
|
||||
if not os.environ.get("CLAUDE_CODE_OAUTH_TOKEN"):
|
||||
log(
|
||||
"FATAL: CLAUDE_CODE_OAUTH_TOKEN not set (source ~/secrev.env). Refusing to run."
|
||||
)
|
||||
sys.exit(2)
|
||||
|
||||
sys.exit(asyncio.run(main_async(args)))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in a new issue