Consolidates the 18 leaf modules from the r720-plane2-scaffold workflow onto the foundation commit. Full suite: 535 passed, 1 skipped; ruff + format clean. Built (pre-deployment scaffold only — nothing provisioned/enabled): - LangGraph pipeline graph.py (INTAKE->CLARIFY->PLAN, interrupt()/resume, checkpointer-injectable) - nodes: clarifier (98% gate), planner, review_loop (GPT-4.1), builders->candidate diff, verifier - §3.3.1 HITL: ledger ops, resume_worker, deadline_timer, recovery sweep, responder - transports: slack / github / claude_code adapters - ci_gate (pure-code pass/fail), operator_cli, run-team.py entry, P1 sim harness - ci/agent-team-apply-verify.yml (split untrusted/privileged jobs) — authored, disabled KNOWN OPEN FINDINGS (verifier/cross-review, not yet fixed — see follow-up): - builders denylist: 4 execution-proven bypasses (delete, mode-change, copy-to, out-of-scope delete) - §3.3.1 CAS: BEGIN IMMEDIATE outside try/except; shared-connection txn nesting unsafe under concurrency - operator_cli: missing re-deliver/force-resume; audit-after-mutate ordering gap - ci yaml: GPT-4.1 cross-review PASS w/ 4 FIX items (symlink path escape, etc.) - P1 sim harness models the ledger layer, not real LangGraph interrupt/resume; P1 exit criteria not yet truly proven Deploy-gated (NOT done): IAM/step-ca/Roles Anywhere/confluence-bot provisioning, /sh-security-review sign-off, live Slack/CI, rsync, live dry-runs, Adam approval.
253 lines
9.7 KiB
Python
253 lines
9.7 KiB
Python
"""Verifier node — the VERIFY-stage LangGraph node (design §3.3, §3.3.2 P3).
|
|
|
|
The verifier reads the authenticated CI results for a task's candidate diff and
|
|
either advances it to a draft PR or loops it back to the builders. The single
|
|
load-bearing rule from §3.3.2 boundary #4 is enforced structurally here:
|
|
|
|
**The verifier *agent* cannot declare success.** Pass/fail is owned by the
|
|
pure-code gate (:mod:`agent_team.ci_gate`) over the authenticated,
|
|
patch-independent CI conclusion. The LLM verifier only ever reads *failures*
|
|
to propose the next fix.
|
|
|
|
So this node calls :func:`agent_team.ci_gate.evaluate_ci_gate` for the decision
|
|
and consults an (optional, injectable) LLM **only** on a FAIL/BLOCK to author a
|
|
fix hint for the builders. The LLM is never asked whether the task passed.
|
|
|
|
State contract (mirrors :class:`agent_team.task_model.PipelineState`):
|
|
|
|
* reads ``candidate_diff``, ``diff_hash`` (ledger hash), ``ci_results``;
|
|
* writes ``status``, ``current_phase``, ``review_verdicts`` (appends the gate
|
|
verdict), and ``ci_results`` (annotated with the gate decision for
|
|
provenance).
|
|
|
|
Transitions (the §3.3 "Stability + autonomy bounds" — the verifier must pass or
|
|
the task loops/holds, never ships):
|
|
|
|
* gate PASS -> :class:`~agent_team.task_model.Phase.DONE`,
|
|
:class:`~agent_team.task_model.TaskStatus.DONE` (verifier produced a draft PR).
|
|
* gate FAIL -> :class:`~agent_team.task_model.Phase.BUILD`,
|
|
:class:`~agent_team.task_model.TaskStatus.ACTIVE` (loop back to builders),
|
|
unless the per-task build-loop budget is spent, in which case it PARKs.
|
|
* gate BLOCK -> :class:`~agent_team.task_model.Phase.PARKED`,
|
|
:class:`~agent_team.task_model.TaskStatus.PARKED` — an ALARM-worthy trust
|
|
violation (denylist hit, hash mismatch, ambiguous conclusion) is never auto-
|
|
retried; it parks for human + GPT cross-review.
|
|
|
|
This module imports the committed foundation contracts verbatim and redefines
|
|
none of them. The LLM seam is injectable (mirroring
|
|
:func:`agent_team.billing.set_invoker`) so the node is pure-function testable
|
|
with no SDK or network.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timezone
|
|
from typing import Any, Callable, Mapping
|
|
|
|
from agent_team.ci_gate import GateDecision, GateResult, evaluate_ci_gate
|
|
from agent_team.task_model import Phase, PipelineState, TaskStatus
|
|
|
|
__all__ = [
|
|
"DEFAULT_MAX_BUILD_LOOPS",
|
|
"VerifierConfig",
|
|
"set_fix_advisor",
|
|
"verifier_node",
|
|
]
|
|
|
|
# Default cap on build<->verify loops before a still-failing task parks rather
|
|
# than spinning (§3.3 bound #6 "N failed build loops"). Configurable per task
|
|
# via :class:`VerifierConfig`.
|
|
DEFAULT_MAX_BUILD_LOOPS: int = 3
|
|
|
|
|
|
@dataclass
|
|
class VerifierConfig:
|
|
"""Per-invocation knobs for the verifier node (§3.3, §3.3.2).
|
|
|
|
``expected_run_id`` keys the gate to the exact CI run the verifier
|
|
dispatched for this diff (a stale/substituted run id is a BLOCK).
|
|
``allowed_scope`` is the task's declared-scope path prefixes for the
|
|
denylist boundary. ``max_build_loops`` caps build<->verify retries before
|
|
the task parks. ``build_loops`` is the loops already consumed for this task
|
|
(the coordinator threads it through state).
|
|
"""
|
|
|
|
expected_run_id: str
|
|
allowed_scope: list[str] | None = None
|
|
max_build_loops: int = DEFAULT_MAX_BUILD_LOOPS
|
|
build_loops: int = 0
|
|
|
|
|
|
# Fix-advisor seam: signature (gate_result, state) -> str. Consulted ONLY on a
|
|
# gate FAIL/BLOCK to author a fix hint for the builders from the *failure*
|
|
# details. Never consulted on PASS — the gate, not the LLM, owns success. The
|
|
# default returns no hint so an un-wired environment degrades to "loop back with
|
|
# no extra guidance" rather than crashing.
|
|
FixAdvisor = Callable[[GateResult, Mapping[str, Any]], str]
|
|
|
|
|
|
def _null_advisor(gate_result: GateResult, state: Mapping[str, Any]) -> str:
|
|
return ""
|
|
|
|
|
|
_fix_advisor: FixAdvisor = _null_advisor
|
|
|
|
|
|
def set_fix_advisor(advisor: FixAdvisor) -> None:
|
|
"""Bind the LLM fix-advisor consulted on a gate FAIL/BLOCK.
|
|
|
|
Leaves call this once at startup with an implementation that reads the gate
|
|
failure reasons + the CI logs and authors a next-fix hint for the builders.
|
|
Keeping it injectable keeps this node dependency-free and unit-testable, and
|
|
structurally enforces that the advisor is only ever invoked on failure (this
|
|
module never calls it on a PASS).
|
|
"""
|
|
global _fix_advisor
|
|
_fix_advisor = advisor
|
|
|
|
|
|
def _utc_now_iso() -> str:
|
|
return datetime.now(timezone.utc).isoformat()
|
|
|
|
|
|
def _verdict(
|
|
gate_result: GateResult,
|
|
*,
|
|
next_phase: Phase,
|
|
build_loops: int,
|
|
fix_hint: str = "",
|
|
) -> dict[str, Any]:
|
|
"""Build the review-verdict record appended to ``review_verdicts``."""
|
|
return {
|
|
"stage": "verify",
|
|
"decision": gate_result.decision.value,
|
|
"reasons": list(gate_result.reasons),
|
|
"run_id": gate_result.run_id,
|
|
"diff_hash": gate_result.diff_hash,
|
|
"ci_conclusion": gate_result.ci_conclusion,
|
|
"next_phase": next_phase.value,
|
|
"build_loops": build_loops,
|
|
"fix_hint": fix_hint,
|
|
"at": _utc_now_iso(),
|
|
}
|
|
|
|
|
|
def verifier_node(
|
|
state: PipelineState,
|
|
config: VerifierConfig,
|
|
) -> PipelineState:
|
|
"""Run the VERIFY stage: gate the CI result, decide the next phase.
|
|
|
|
Pure function of ``state`` + ``config`` (the gate does no I/O; the optional
|
|
fix-advisor is the only outbound call, and only on failure). Returns a
|
|
**partial** :class:`PipelineState` update (``total=False``) carrying the new
|
|
``status``/``current_phase``, an appended ``review_verdicts`` entry, and a
|
|
gate-annotated ``ci_results`` for provenance — the LangGraph reducer merges
|
|
it into the durable thread state.
|
|
|
|
Decision flow (§3.3.2 boundary #4 + §3.3 autonomy bounds):
|
|
|
|
1. Call :func:`agent_team.ci_gate.evaluate_ci_gate` with the candidate diff,
|
|
the ledger hash (``diff_hash``), the authenticated ``ci_results``, the
|
|
``expected_run_id``, and the task's ``allowed_scope``.
|
|
2. PASS -> advance to DONE (draft PR). The advisor is NOT consulted.
|
|
3. FAIL -> consult the fix-advisor for a hint, then loop back to BUILD —
|
|
unless ``build_loops`` has reached ``max_build_loops``, in which case
|
|
PARK (a task that keeps failing must hold, never ship: §3.3 bound #3/#6).
|
|
4. BLOCK -> PARK + (advisor hint) for human + GPT cross-review. A trust
|
|
violation is never auto-retried.
|
|
"""
|
|
candidate_diff = state.get("candidate_diff")
|
|
ledger_hash = state.get("diff_hash")
|
|
ci_results = state.get("ci_results")
|
|
|
|
if not isinstance(candidate_diff, str):
|
|
# No diff to verify is itself a refuse-to-proceed: park for a human
|
|
# rather than declaring anything. (A builder must have produced a diff
|
|
# before VERIFY runs.)
|
|
gate_result = GateResult(
|
|
decision=GateDecision.BLOCK,
|
|
reasons=["no candidate_diff present in state to verify"],
|
|
run_id=config.expected_run_id,
|
|
diff_hash=ledger_hash,
|
|
ci_conclusion=None,
|
|
)
|
|
else:
|
|
gate_result = evaluate_ci_gate(
|
|
candidate_diff=candidate_diff,
|
|
ledger_hash=ledger_hash,
|
|
ci_result=ci_results,
|
|
expected_run_id=config.expected_run_id,
|
|
allowed_scope=config.allowed_scope,
|
|
)
|
|
|
|
annotated_ci = dict(ci_results) if isinstance(ci_results, Mapping) else {}
|
|
annotated_ci["gate_decision"] = gate_result.decision.value
|
|
annotated_ci["gate_reasons"] = list(gate_result.reasons)
|
|
|
|
if gate_result.decision is GateDecision.PASS:
|
|
verdict = _verdict(
|
|
gate_result, next_phase=Phase.DONE, build_loops=config.build_loops
|
|
)
|
|
return {
|
|
"status": TaskStatus.DONE.value,
|
|
"current_phase": Phase.DONE.value,
|
|
"review_verdicts": [verdict],
|
|
"ci_results": annotated_ci,
|
|
"updated_at": verdict["at"],
|
|
}
|
|
|
|
# Every non-PASS path consults the fix-advisor on the *failure* details.
|
|
# This is the only place the LLM is touched, and it never decides success.
|
|
fix_hint = _fix_advisor(gate_result, state)
|
|
|
|
if gate_result.decision is GateDecision.FAIL:
|
|
next_loops = config.build_loops + 1
|
|
if next_loops >= config.max_build_loops:
|
|
# Exhausted the build-loop budget: hold rather than spin (§3.3 #6).
|
|
verdict = _verdict(
|
|
gate_result,
|
|
next_phase=Phase.PARKED,
|
|
build_loops=next_loops,
|
|
fix_hint=fix_hint,
|
|
)
|
|
verdict["reasons"].append(
|
|
f"max build loops ({config.max_build_loops}) reached; parking"
|
|
)
|
|
return {
|
|
"status": TaskStatus.PARKED.value,
|
|
"current_phase": Phase.PARKED.value,
|
|
"review_verdicts": [verdict],
|
|
"ci_results": annotated_ci,
|
|
"updated_at": verdict["at"],
|
|
}
|
|
verdict = _verdict(
|
|
gate_result,
|
|
next_phase=Phase.BUILD,
|
|
build_loops=next_loops,
|
|
fix_hint=fix_hint,
|
|
)
|
|
return {
|
|
"status": TaskStatus.ACTIVE.value,
|
|
"current_phase": Phase.BUILD.value,
|
|
"review_verdicts": [verdict],
|
|
"ci_results": annotated_ci,
|
|
"updated_at": verdict["at"],
|
|
}
|
|
|
|
# GateDecision.BLOCK — trust violation. Park for human + GPT cross-review;
|
|
# never auto-retried, never shipped.
|
|
verdict = _verdict(
|
|
gate_result,
|
|
next_phase=Phase.PARKED,
|
|
build_loops=config.build_loops,
|
|
fix_hint=fix_hint,
|
|
)
|
|
return {
|
|
"status": TaskStatus.PARKED.value,
|
|
"current_phase": Phase.PARKED.value,
|
|
"review_verdicts": [verdict],
|
|
"ci_results": annotated_ci,
|
|
"updated_at": verdict["at"],
|
|
}
|