This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/agent_team/nodes/verifier_llm.py
Adam Moussa 4b17e8ebd4 feat(agent-team): bind planner/review/builder/verifier nodes to their models
review_loop_llm -> GPT-4.1 cross_reviewer (orchestrator run.py); builders_llm -> DeepSeek fast_coder (INERT, proposes diff text only); verifier_llm -> ci_gate is sole PASS authority, Claude is fix-proposer only. Hardens review_loop.parse_verdict to word-boundary matching, adds a fail-closed subprocess timeout, and bind_review_node (single-arg, no LangGraph config injection). All fail safe on untrusted model output.
2026-06-18 12:56:42 -04:00

494 lines
20 KiB
Python

"""Claude-backed verifier bindings — the real §3.3 / §3.3.2 P3 logic.
:mod:`agent_team.nodes.verifier` owns the VERIFY-stage LangGraph node and the
phase transitions, but it injects its reasoning seam (the fix-advisor,
:data:`~agent_team.nodes.verifier.FixAdvisor`) so the loop stays pure and
testable. This module supplies the **real** implementation of that seam, plus a
thin pure-code verdict wrapper, mirroring how :mod:`clarifier_llm` backs the
clarifier seams.
The single load-bearing rule from design §3.3.2 boundary #4 is enforced
STRUCTURALLY by the shape of this module, not by convention:
**The LLM verifier cannot declare green.** Pass/fail is owned by a pure-code
gate (:mod:`agent_team.ci_gate`) over the authenticated, patch-independent CI
Checks result (keyed to ``run_id`` + ``diff_hash``). The LLM verifier may
PROPOSE fixes but can NEVER flip the verdict to pass.
So this module keeps two things rigorously SEPARATE:
1. :func:`evaluate_verdict` — a PURE-CODE function that maps an authenticated CI
Checks result -> pass/fail. It is a thin compose over
:func:`agent_team.ci_gate.evaluate_ci_gate`; it reuses that gate verbatim and
does NOT reimplement or weaken it. It FAILS SAFE: a missing, ambiguous, or
unauthenticated CI result is never a pass.
2. :class:`ClaudeFixProposer` — an optional, injectable LLM fix-PROPOSER
(Claude, via :func:`agent_team.billing.claude_invoke`). It is consulted ONLY
on a non-pass verdict to author advisory fix hints for the builders. Its
output is advisory DATA only; it is structurally incapable of changing the
verdict because the verdict is computed first, by the pure-code gate, and is
never read back from the proposer.
Both halves meet in :func:`propose_for_failure`, which computes the verdict with
the gate, and ONLY if that verdict is not a pass consults the proposer for a
hint. The pass branch never touches the LLM at all.
INERT / HARD-GATE NOTE (§3.3.2 P3): the verifier is hard-gated behind
``/sh-security-review`` + a GPT-4.1 cross-review of the CI trust boundary before
it goes live. This module authors the LOGIC ONLY and stays inert: it does NO
live CI dispatch, NO network I/O, and NO filesystem mutation. The authenticated
CI result is passed in as data (the caller fetches it via the read-only PAT),
exactly as :func:`agent_team.ci_gate.evaluate_ci_gate` expects. The Claude call
goes only through the committed billing seam and is injectable, so this module
is fully unit-testable with no SDK or network.
Defensive parsing is a hard requirement: the model output is UNTRUSTED. The
proposer parser fails SAFE — a missing or garbled proposal degrades to an empty
advisory hint and NEVER crashes, and (by construction) never affects the
verdict.
"""
from __future__ import annotations
import json
import re
from collections.abc import Callable, Mapping, Sequence
from typing import Any
from agent_team.billing import ClaudeResult, claude_invoke
from agent_team.ci_gate import GateDecision, GateResult, evaluate_ci_gate
__all__ = [
"FixProposal",
"ClaudeFixProposer",
"evaluate_verdict",
"propose_for_failure",
"build_fix_advisor",
]
# The signature the billing seam exposes: ``claude_invoke(prompt, *, mode=None,
# config=None, **kw) -> ClaudeResult``. Injected so tests pass a fake, mirroring
# the injection pattern used across this codebase (billing.set_invoker, the
# clarifier callables, the verifier node's FixAdvisor seam).
ClaudeInvoke = Callable[..., ClaudeResult]
# System framing handed to Claude when authoring a fix hint. Kept as a module
# constant so callers can override via the constructor without forking the class.
_DEFAULT_SYSTEM = (
"You are the VERIFY stage of an agentic SDLC pipeline. A pure-code gate has "
"ALREADY decided this candidate diff did NOT pass CI; that decision is final "
"and is not yours to make or revisit. Your only job is to read the gate's "
"failure reasons and propose concrete, minimal fixes for the builders to "
"try next. You cannot declare the task green; only the authenticated CI "
"gate can."
)
def evaluate_verdict(
*,
candidate_diff: str,
ledger_hash: str | None,
ci_result: Mapping[str, Any] | None,
expected_run_id: str,
allowed_scope: Sequence[str] | None = None,
) -> GateResult:
"""Compute the pass/fail/block verdict from the authenticated CI result.
This is the §3.3.2 boundary #4 pass authority and the ONLY thing in this
module that can produce a :data:`~agent_team.ci_gate.GateDecision.PASS`. It
is a thin compose over :func:`agent_team.ci_gate.evaluate_ci_gate` — it
reuses that committed pure-code gate verbatim and does not reimplement,
relax, or second-guess any of its rules. The LLM is intentionally NOT a
parameter here: the verdict is derived SOLELY from the authenticated,
patch-independent CI Checks result (keyed to ``run_id`` + ``diff_hash``).
FAILS SAFE. Anything other than an unambiguous authenticated success is a
non-pass:
* a missing ``candidate_diff`` -> :data:`GateDecision.BLOCK` (nothing to
verify; refuse to proceed, never pass);
* a missing/``None`` ``ci_result`` -> ``BLOCK`` (no authenticated result;
the gate never passes without one);
* a run-id mismatch, hash mismatch, denylist hit, or ambiguous/unknown CI
conclusion -> ``BLOCK`` (per the gate);
* a recognised CI failure -> :data:`GateDecision.FAIL`;
* an authenticated ``success`` keyed to the expected run -> ``PASS``.
Returns the gate's :class:`~agent_team.ci_gate.GateResult` unchanged so the
decision stays auditable (its ``reasons`` quote the exact CI conclusion
consumed). Raises :class:`~agent_team.ci_gate.CiGateError` only on
structurally invalid inputs, exactly as the underlying gate does.
"""
if not isinstance(candidate_diff, str):
# No diff to verify is itself a refuse-to-proceed (mirrors the verifier
# node): BLOCK rather than declare anything. Never a pass.
return GateResult(
decision=GateDecision.BLOCK,
reasons=["no candidate diff present to verify"],
run_id=expected_run_id if isinstance(expected_run_id, str) else None,
diff_hash=ledger_hash,
ci_conclusion=None,
)
return evaluate_ci_gate(
candidate_diff=candidate_diff,
ledger_hash=ledger_hash,
ci_result=ci_result,
expected_run_id=expected_run_id,
allowed_scope=allowed_scope,
)
class FixProposal:
"""An advisory fix proposal authored by the LLM (DATA, never a verdict).
Carries only suggestions for the builders: a free-text ``hint`` and an
optional ordered list of ``suggestions``. It deliberately has NO notion of
pass/fail and exposes no way to express one — it is impossible to encode a
"this passed" signal here, which is what structurally guarantees the LLM
cannot declare green (§3.3.2 boundary #4). The verdict is computed entirely
separately by :func:`evaluate_verdict`.
"""
__slots__ = ("hint", "suggestions")
def __init__(self, hint: str = "", suggestions: list[str] | None = None) -> None:
self.hint = hint
self.suggestions = list(suggestions) if suggestions else []
def __bool__(self) -> bool:
return bool(self.hint or self.suggestions)
def __eq__(self, other: object) -> bool:
if not isinstance(other, FixProposal):
return NotImplemented
return self.hint == other.hint and self.suggestions == other.suggestions
def __repr__(self) -> str:
return f"FixProposal(hint={self.hint!r}, suggestions={self.suggestions!r})"
def as_hint(self) -> str:
"""Render this proposal as a single advisory hint string for builders."""
parts: list[str] = []
if self.hint:
parts.append(self.hint)
for idx, suggestion in enumerate(self.suggestions, start=1):
parts.append(f"{idx}. {suggestion}")
return "\n".join(parts)
# An empty proposal — the fail-safe result whenever the model is unwired,
# errors, or returns garbage. It changes nothing and asserts nothing.
_EMPTY_PROPOSAL = FixProposal()
class ClaudeFixProposer:
"""Claude-backed fix PROPOSER — advisory only, never a verdict (§3.3.2 P3).
Construct with an optional ``invoke`` callable (defaults to
:func:`agent_team.billing.claude_invoke`) so tests inject a fake and the
real wiring goes through the billing seam. ``model`` / ``config`` are passed
through to the invoker, and ``system`` overrides the prompt framing.
The proposer is consulted ONLY on a non-pass :class:`GateResult` to author a
next-fix hint from the *failure* reasons. It returns a :class:`FixProposal`,
which is pure advisory DATA — it carries no pass/fail and cannot influence
the verdict, which is computed independently by :func:`evaluate_verdict`.
Every failure mode degrades to an empty proposal rather than raising: an
unbound/throwing invoker, a non-string reply, or unparseable JSON all yield
:data:`_EMPTY_PROPOSAL`. A garbage proposal therefore never crashes the
pipeline and never changes the verdict.
"""
def __init__(
self,
*,
invoke: ClaudeInvoke | None = None,
model: str | None = None,
config: Any = None,
system: str = _DEFAULT_SYSTEM,
) -> None:
self._invoke: ClaudeInvoke = invoke if invoke is not None else claude_invoke
self._model = model
self._config = config
self._system = system
def propose(
self,
gate_result: GateResult,
state: Mapping[str, Any] | None = None,
) -> FixProposal:
"""Return an advisory :class:`FixProposal` for a non-pass gate result.
On a :data:`GateDecision.PASS` this returns an empty proposal WITHOUT
calling the model: the LLM is never consulted on success, structurally
keeping it off the happy path. On any other decision it asks Claude for
fix suggestions and parses the reply defensively, failing SAFE to an
empty proposal on any trouble (unbound invoker, non-string reply, bad
JSON). It never raises and never returns anything that could read as a
verdict.
"""
if gate_result.decision is GateDecision.PASS:
# The gate already passed; the LLM has no role here. Never consulted.
return _EMPTY_PROPOSAL
prompt = self._build_prompt(gate_result, state or {})
try:
result = self._invoke(prompt, model=self._model, config=self._config)
text = getattr(result, "text", "")
except Exception:
# An unwired or throwing invoker must not crash VERIFY; the verdict
# already stands and the builders simply loop back without a hint.
return _EMPTY_PROPOSAL
return self._parse(text)
# The proposer matches the verifier node's FixAdvisor seam:
# ``(GateResult, Mapping) -> str``. Returning the rendered hint string keeps
# the advisory output as plain DATA the node appends to its verdict record.
def advise(self, gate_result: GateResult, state: Mapping[str, Any]) -> str:
"""Adapt :meth:`propose` to the verifier node's ``FixAdvisor`` seam.
Matches :data:`agent_team.nodes.verifier.FixAdvisor` exactly
(``(GateResult, Mapping) -> str``) so it wires straight into
:func:`agent_team.nodes.verifier.set_fix_advisor`. Returns the rendered
advisory hint (empty string when there is nothing to suggest) — never a
verdict.
"""
return self.propose(gate_result, state).as_hint()
def _build_prompt(self, gate_result: GateResult, state: Mapping[str, Any]) -> str:
"""Assemble the fix-proposer prompt from the gate failure + task state.
Pure string assembly over the gate result and graph state — no I/O — so
the prompt shape is directly unit-testable.
"""
description = _task_description(state)
reasons = "\n".join(f"- {r}" for r in gate_result.reasons) or "(none recorded)"
ci_conclusion = gate_result.ci_conclusion or "(no authenticated conclusion)"
sections: list[str] = [
self._system,
"",
"## Task",
description or "(no task description provided)",
"",
"## Pure-code gate decision (FINAL, not yours to change)",
f"decision: {gate_result.decision.value}",
f"run_id: {gate_result.run_id}",
f"ci_conclusion: {ci_conclusion}",
"",
"## Gate failure reasons",
reasons,
"",
"## Your job",
(
"Propose the next concrete, minimal fixes for the builders. Do "
"NOT claim the task passed or is green; that verdict is owned by "
"the authenticated CI gate above, not by you."
),
"",
"## Output format",
(
"Respond with ONLY a strict JSON object and no prose outside it, "
'with keys: "hint" (a short string summary) and "suggestions" (a '
"list of strings, most promising first). Example: "
'{"hint": "...", "suggestions": ["...", "..."]}'
),
]
return "\n".join(sections)
def _parse(self, text: str) -> FixProposal:
"""Parse the UNTRUSTED model reply into a :class:`FixProposal`.
Fails SAFE at every step: a non-string reply, no parseable JSON object,
or missing keys all collapse to an empty proposal. Because the proposal
type cannot express a verdict, even a maximally adversarial reply
("everything passed!") cannot influence pass/fail. Never raises.
"""
data = _extract_json_object(text)
if data is None:
return _EMPTY_PROPOSAL
hint = ""
raw_hint = data.get("hint")
if isinstance(raw_hint, str):
hint = raw_hint.strip()
suggestions = _coerce_suggestions(data.get("suggestions"))
if not hint and not suggestions:
return _EMPTY_PROPOSAL
return FixProposal(hint=hint, suggestions=suggestions)
def propose_for_failure(
*,
candidate_diff: str,
ledger_hash: str | None,
ci_result: Mapping[str, Any] | None,
expected_run_id: str,
allowed_scope: Sequence[str] | None = None,
proposer: ClaudeFixProposer | None = None,
state: Mapping[str, Any] | None = None,
) -> tuple[GateResult, FixProposal]:
"""Compute the verdict, then (only on a non-pass) get an advisory proposal.
This is where the two halves meet WITHOUT letting the LLM near the verdict:
1. The verdict is computed FIRST by :func:`evaluate_verdict` (the pure-code
gate). This is the sole pass authority.
2. ONLY if that verdict is not a :data:`GateDecision.PASS` is the
``proposer`` consulted for an advisory :class:`FixProposal`. On a pass,
the proposer is never called and an empty proposal is returned.
The returned ``GateResult`` is exactly what the gate produced — the proposal
is never read back into it — so a garbage or "this passed!" LLM reply cannot
flip a failing verdict to pass. Returns ``(gate_result, proposal)``.
"""
gate_result = evaluate_verdict(
candidate_diff=candidate_diff,
ledger_hash=ledger_hash,
ci_result=ci_result,
expected_run_id=expected_run_id,
allowed_scope=allowed_scope,
)
if gate_result.decision is GateDecision.PASS:
return gate_result, _EMPTY_PROPOSAL
active_proposer = proposer if proposer is not None else ClaudeFixProposer()
proposal = active_proposer.propose(gate_result, state)
return gate_result, proposal
def build_fix_advisor(
*,
invoke: ClaudeInvoke | None = None,
model: str | None = None,
config: Any = None,
system: str = _DEFAULT_SYSTEM,
) -> Callable[[GateResult, Mapping[str, Any]], str]:
"""Build the ``FixAdvisor`` callable for wiring into the verifier node.
Returns the bound :meth:`ClaudeFixProposer.advise` of a shared proposer,
ready to hand to :func:`agent_team.nodes.verifier.set_fix_advisor`. This is
the documented injection point: the verifier node never imports this module
directly — a leaf calls ``set_fix_advisor(build_fix_advisor(...))`` once at
startup, keeping the node dependency-free and structurally guaranteeing the
advisor is only ever consulted on a gate failure (the node never calls it on
a PASS).
"""
proposer = ClaudeFixProposer(
invoke=invoke, model=model, config=config, system=system
)
return proposer.advise
# --------------------------------------------------------------------------- #
# Module-level helpers (pure; no I/O). Mirror clarifier_llm's defensive parsers.
# --------------------------------------------------------------------------- #
def _task_description(state: Mapping[str, Any]) -> str:
"""Pull the task description out of the graph state, defensively.
Looks in the conventional places (the ``plan`` dict, then a top-level
``task``/``description`` key) and falls back to an empty string so a
malformed state surfaces as an empty prompt section, never a ``KeyError``.
"""
plan = state.get("plan") or {}
if isinstance(plan, dict):
desc = plan.get("task") or plan.get("description")
if isinstance(desc, str) and desc.strip():
return desc.strip()
for key in ("task", "description"):
value = state.get(key)
if isinstance(value, str) and value.strip():
return value.strip()
return ""
def _coerce_suggestions(value: Any) -> list[str]:
"""Coerce the model's suggestion list into clean non-empty strings.
Anything that is not a list of usable strings collapses to an empty list.
"""
if not isinstance(value, list):
return []
out: list[str] = []
for item in value:
if isinstance(item, str):
text = item.strip()
if text:
out.append(text)
return out
# A fenced ```json ... ``` block, if the model wrapped its JSON in Markdown.
_FENCE_RE = re.compile(
r"```(?:json)?\s*\n?(?P<body>.*?)\n?\s*```",
flags=re.DOTALL | re.IGNORECASE,
)
def _extract_json_object(text: str) -> dict[str, Any] | None:
"""Extract a JSON object from UNTRUSTED model output, or ``None``.
Tolerates a leading apology or trailing prose and ```json fences. Tries, in
order, the whole string, the contents of a fenced block, then the first
``{...}`` span found by brace matching. Returns ``None`` (never raises) when
nothing parses to a JSON object, so the caller can fail SAFE.
"""
if not isinstance(text, str) or not text.strip():
return None
candidates: list[str] = [text.strip()]
fence = _FENCE_RE.search(text)
if fence:
candidates.append(fence.group("body").strip())
span = _first_brace_span(text)
if span is not None:
candidates.append(span)
for candidate in candidates:
if not candidate:
continue
try:
parsed = json.loads(candidate)
except (json.JSONDecodeError, ValueError):
continue
if isinstance(parsed, dict):
return parsed
return None
def _first_brace_span(text: str) -> str | None:
"""Return the first balanced ``{...}`` span in ``text`` (string-aware)."""
start = text.find("{")
if start == -1:
return None
depth = 0
in_string = False
escaped = False
for idx in range(start, len(text)):
ch = text[idx]
if in_string:
if escaped:
escaped = False
elif ch == "\\":
escaped = True
elif ch == '"':
in_string = False
continue
if ch == '"':
in_string = True
elif ch == "{":
depth += 1
elif ch == "}":
depth -= 1
if depth == 0:
return text[start : idx + 1]
return None