Consolidates the 18 leaf modules from the r720-plane2-scaffold workflow onto the foundation commit. Full suite: 535 passed, 1 skipped; ruff + format clean. Built (pre-deployment scaffold only — nothing provisioned/enabled): - LangGraph pipeline graph.py (INTAKE->CLARIFY->PLAN, interrupt()/resume, checkpointer-injectable) - nodes: clarifier (98% gate), planner, review_loop (GPT-4.1), builders->candidate diff, verifier - §3.3.1 HITL: ledger ops, resume_worker, deadline_timer, recovery sweep, responder - transports: slack / github / claude_code adapters - ci_gate (pure-code pass/fail), operator_cli, run-team.py entry, P1 sim harness - ci/agent-team-apply-verify.yml (split untrusted/privileged jobs) — authored, disabled KNOWN OPEN FINDINGS (verifier/cross-review, not yet fixed — see follow-up): - builders denylist: 4 execution-proven bypasses (delete, mode-change, copy-to, out-of-scope delete) - §3.3.1 CAS: BEGIN IMMEDIATE outside try/except; shared-connection txn nesting unsafe under concurrency - operator_cli: missing re-deliver/force-resume; audit-after-mutate ordering gap - ci yaml: GPT-4.1 cross-review PASS w/ 4 FIX items (symlink path escape, etc.) - P1 sim harness models the ledger layer, not real LangGraph interrupt/resume; P1 exit criteria not yet truly proven Deploy-gated (NOT done): IAM/step-ca/Roles Anywhere/confluence-bot provisioning, /sh-security-review sign-off, live Slack/CI, rsync, live dry-runs, Adam approval.
204 lines
7.3 KiB
Python
204 lines
7.3 KiB
Python
"""Unit tests for agent_team.nodes.verifier — the VERIFY node (§3.3, §3.3.2)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from agent_team.ci_gate import GateDecision, GateResult
|
|
from agent_team.nodes import verifier as verifier_mod
|
|
from agent_team.nodes.verifier import (
|
|
DEFAULT_MAX_BUILD_LOOPS,
|
|
VerifierConfig,
|
|
set_fix_advisor,
|
|
verifier_node,
|
|
)
|
|
from agent_team.state_store import compute_content_hash
|
|
from agent_team.task_model import Phase, PipelineState, TaskStatus
|
|
|
|
|
|
def _diff_for(*paths: str) -> str:
|
|
chunks = []
|
|
for p in paths:
|
|
chunks.append(f"diff --git a/{p} b/{p}\n@@ -1 +1 @@\n-old\n+new\n")
|
|
return "".join(chunks)
|
|
|
|
|
|
def _hash(diff: str) -> str:
|
|
return compute_content_hash(diff.encode("utf-8"))
|
|
|
|
|
|
def _state(diff: str, ci: dict | None) -> PipelineState:
|
|
return {
|
|
"thread_id": "t1",
|
|
"status": TaskStatus.ACTIVE.value,
|
|
"current_phase": Phase.VERIFY.value,
|
|
"candidate_diff": diff,
|
|
"diff_hash": _hash(diff),
|
|
"ci_results": ci,
|
|
}
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _reset_advisor():
|
|
"""Restore the default null advisor after each test."""
|
|
yield
|
|
set_fix_advisor(verifier_mod._null_advisor)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# PASS path
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def test_pass_advances_to_done() -> None:
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r1", "conclusion": "success", "diff_hash": _hash(diff)}
|
|
out = verifier_node(_state(diff, ci), VerifierConfig(expected_run_id="r1"))
|
|
assert out["status"] == TaskStatus.DONE.value
|
|
assert out["current_phase"] == Phase.DONE.value
|
|
assert out["ci_results"]["gate_decision"] == "pass"
|
|
|
|
|
|
def test_pass_does_not_consult_advisor() -> None:
|
|
calls: list = []
|
|
|
|
def advisor(result, state): # pragma: no cover - asserted not called
|
|
calls.append(result)
|
|
return "hint"
|
|
|
|
set_fix_advisor(advisor)
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r1", "conclusion": "success", "diff_hash": _hash(diff)}
|
|
verifier_node(_state(diff, ci), VerifierConfig(expected_run_id="r1"))
|
|
assert calls == [] # the LLM is never asked whether it passed
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# FAIL path — loop back to BUILD
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def test_fail_loops_back_to_build() -> None:
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r1", "conclusion": "failure", "diff_hash": _hash(diff)}
|
|
out = verifier_node(
|
|
_state(diff, ci), VerifierConfig(expected_run_id="r1", build_loops=0)
|
|
)
|
|
assert out["status"] == TaskStatus.ACTIVE.value
|
|
assert out["current_phase"] == Phase.BUILD.value
|
|
|
|
|
|
def test_fail_consults_advisor_for_hint() -> None:
|
|
def advisor(result: GateResult, state) -> str:
|
|
assert result.decision is GateDecision.FAIL
|
|
return "bump the pinned version"
|
|
|
|
set_fix_advisor(advisor)
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r1", "conclusion": "failure", "diff_hash": _hash(diff)}
|
|
out = verifier_node(_state(diff, ci), VerifierConfig(expected_run_id="r1"))
|
|
assert out["review_verdicts"][0]["fix_hint"] == "bump the pinned version"
|
|
|
|
|
|
def test_fail_parks_when_build_loops_exhausted() -> None:
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r1", "conclusion": "failure", "diff_hash": _hash(diff)}
|
|
cfg = VerifierConfig(
|
|
expected_run_id="r1",
|
|
build_loops=DEFAULT_MAX_BUILD_LOOPS - 1,
|
|
max_build_loops=DEFAULT_MAX_BUILD_LOOPS,
|
|
)
|
|
out = verifier_node(_state(diff, ci), cfg)
|
|
assert out["status"] == TaskStatus.PARKED.value
|
|
assert out["current_phase"] == Phase.PARKED.value
|
|
assert any("max build loops" in r for r in out["review_verdicts"][0]["reasons"])
|
|
|
|
|
|
def test_fail_increments_build_loops_in_verdict() -> None:
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r1", "conclusion": "failure", "diff_hash": _hash(diff)}
|
|
out = verifier_node(
|
|
_state(diff, ci), VerifierConfig(expected_run_id="r1", build_loops=1)
|
|
)
|
|
assert out["review_verdicts"][0]["build_loops"] == 2
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# BLOCK path — park for human + GPT cross-review
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def test_block_on_denylist_parks() -> None:
|
|
diff = _diff_for(".github/workflows/ci.yml")
|
|
ci = {"run_id": "r1", "conclusion": "success", "diff_hash": _hash(diff)}
|
|
out = verifier_node(_state(diff, ci), VerifierConfig(expected_run_id="r1"))
|
|
assert out["status"] == TaskStatus.PARKED.value
|
|
assert out["current_phase"] == Phase.PARKED.value
|
|
assert out["ci_results"]["gate_decision"] == "block"
|
|
|
|
|
|
def test_block_on_hash_mismatch_parks() -> None:
|
|
diff = _diff_for("src/foo.py")
|
|
state = _state(diff, {"run_id": "r1", "conclusion": "success"})
|
|
state["diff_hash"] = "tampered"
|
|
out = verifier_node(state, VerifierConfig(expected_run_id="r1"))
|
|
assert out["status"] == TaskStatus.PARKED.value
|
|
|
|
|
|
def test_block_never_advances_to_done_even_with_advisor() -> None:
|
|
set_fix_advisor(lambda result, state: "irrelevant")
|
|
diff = _diff_for(".github/workflows/ci.yml")
|
|
ci = {"run_id": "r1", "conclusion": "success", "diff_hash": _hash(diff)}
|
|
out = verifier_node(_state(diff, ci), VerifierConfig(expected_run_id="r1"))
|
|
assert out["status"] != TaskStatus.DONE.value
|
|
|
|
|
|
def test_missing_diff_parks() -> None:
|
|
state: PipelineState = {
|
|
"thread_id": "t1",
|
|
"candidate_diff": None,
|
|
"diff_hash": None,
|
|
"ci_results": None,
|
|
}
|
|
out = verifier_node(state, VerifierConfig(expected_run_id="r1"))
|
|
assert out["status"] == TaskStatus.PARKED.value
|
|
assert any("no candidate_diff" in r for r in out["review_verdicts"][0]["reasons"])
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# Provenance / partial-update shape
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def test_verdict_records_provenance() -> None:
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r9", "conclusion": "success", "diff_hash": _hash(diff)}
|
|
out = verifier_node(_state(diff, ci), VerifierConfig(expected_run_id="r9"))
|
|
verdict = out["review_verdicts"][0]
|
|
assert verdict["stage"] == "verify"
|
|
assert verdict["run_id"] == "r9"
|
|
assert verdict["diff_hash"] == _hash(diff)
|
|
assert verdict["ci_conclusion"] == "success"
|
|
assert "at" in verdict
|
|
|
|
|
|
def test_returns_partial_update_only() -> None:
|
|
diff = _diff_for("src/foo.py")
|
|
ci = {"run_id": "r1", "conclusion": "success", "diff_hash": _hash(diff)}
|
|
out = verifier_node(_state(diff, ci), VerifierConfig(expected_run_id="r1"))
|
|
# Node returns only the keys it writes (LangGraph reducer merges the rest).
|
|
assert set(out) == {
|
|
"status",
|
|
"current_phase",
|
|
"review_verdicts",
|
|
"ci_results",
|
|
"updated_at",
|
|
}
|
|
|
|
|
|
def test_allowed_scope_threaded_to_gate() -> None:
|
|
diff = _diff_for("src/foo.py", "elsewhere/bar.py")
|
|
ci = {"run_id": "r1", "conclusion": "success", "diff_hash": _hash(diff)}
|
|
cfg = VerifierConfig(expected_run_id="r1", allowed_scope=["src/"])
|
|
out = verifier_node(_state(diff, ci), cfg)
|
|
assert out["status"] == TaskStatus.PARKED.value # out-of-scope -> BLOCK -> park
|