This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/tests/test_review_loop_llm.py
Adam Moussa 4b17e8ebd4 feat(agent-team): bind planner/review/builder/verifier nodes to their models
review_loop_llm -> GPT-4.1 cross_reviewer (orchestrator run.py); builders_llm -> DeepSeek fast_coder (INERT, proposes diff text only); verifier_llm -> ci_gate is sole PASS authority, Claude is fix-proposer only. Hardens review_loop.parse_verdict to word-boundary matching, adds a fail-closed subprocess timeout, and bind_review_node (single-arg, no LangGraph config injection). All fail safe on untrusted model output.
2026-06-18 12:56:42 -04:00

229 lines
8.7 KiB
Python

"""Unit tests for agent_team.nodes.review_loop_llm (§3.3, §7.1 P2).
The real GPT-4.1 cross-family review binding is exercised with a FAKE review
callable that returns canned verdict text — no network, no subprocess. The
load-bearing properties under test:
* **No orchestrator import at module load.** Importing this module must not pull
in the orchestrator package (``models`` / ``graph``); the default reviewer
shells out / imports lazily.
* **Clear approve -> APPROVE.** An injected fake returning an explicit APPROVE
verdict maps to the node-contract ``ReviewVerdict.APPROVE`` (proceed).
* **Changes requested -> REQUEST_CHANGES.** The loop-back / escalate verdict.
* **Fail SAFE.** A review call that raises, or returns garbage / empty / a
non-string, maps to ``REQUEST_CHANGES`` — never an auto-approve.
* **Routing facts.** The default reviewer resolves the orchestrator root at
``parents[3]`` and a ``run.py`` next to it, and is bound as the default.
"""
from __future__ import annotations
import subprocess
import sys
from typing import Any
import pytest
from agent_team.nodes.review_loop import ReviewVerdict
from agent_team.nodes.review_loop_llm import (
default_plan_reviewer,
make_run_py_invoker,
resolve_orchestrator_root,
resolve_run_py,
review_plan,
)
# --------------------------------------------------------------------------- #
# Fakes / helpers
# --------------------------------------------------------------------------- #
class _FakeReview:
"""A fake plan reviewer that returns canned text and records its calls.
``reply`` is the verdict text returned every call. ``raises`` (if set) is
raised instead, to simulate a failed review call.
"""
def __init__(self, reply: Any = "", *, raises: BaseException | None = None) -> None:
self._reply = reply
self._raises = raises
self.calls: list[dict[str, Any]] = []
def __call__(self, prompt: str, **kw: Any) -> Any:
self.calls.append({"prompt": prompt, "kw": kw})
if self._raises is not None:
raise self._raises
return self._reply
_PLAN = {"task": "ship a thing", "phases": [{"name": "P1"}, {"name": "P2"}]}
def _state() -> dict[str, Any]:
return {"plan": _PLAN, "review_verdicts": []}
# --------------------------------------------------------------------------- #
# Module import hygiene
# --------------------------------------------------------------------------- #
def test_module_imports_without_orchestrator() -> None:
"""Importing the module must not import the orchestrator package."""
# The module is already imported at top, but assert the orchestrator stack
# did not get pulled in as a side effect of importing it.
assert "models" not in sys.modules
assert "graph" not in sys.modules
# --------------------------------------------------------------------------- #
# Verdict mapping
# --------------------------------------------------------------------------- #
def test_clear_approve_maps_to_approve() -> None:
"""An injected fake returning a clear APPROVE verdict -> ReviewVerdict.APPROVE."""
fake = _FakeReview("VERDICT: APPROVE\nThe plan is sound and ready to build.")
verdict = review_plan(_PLAN, _state(), review=fake)
assert verdict is ReviewVerdict.APPROVE
def test_changes_requested_maps_to_request_changes() -> None:
"""An injected fake returning changes-requested -> ReviewVerdict.REQUEST_CHANGES."""
fake = _FakeReview("VERDICT: REQUEST CHANGES\nPhase ordering is wrong.")
verdict = review_plan(_PLAN, _state(), review=fake)
assert verdict is ReviewVerdict.REQUEST_CHANGES
def test_block_token_maps_to_request_changes() -> None:
"""A BLOCK verdict (sh-plan-review vocabulary) -> REQUEST_CHANGES."""
fake = _FakeReview("BLOCK: missing rollback phase.")
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
def test_prompt_only_calling_convention() -> None:
"""review_plan(prompt=...) reviews an already-composed prompt (the seam shape)."""
fake = _FakeReview("APPROVE")
verdict = review_plan(prompt="pre-composed review task", review=fake)
assert verdict is ReviewVerdict.APPROVE
assert fake.calls[0]["prompt"] == "pre-composed review task"
def test_prompt_is_composed_from_plan_when_not_supplied() -> None:
"""With no prompt, the plan text is embedded in the composed review task."""
fake = _FakeReview("APPROVE")
review_plan(_PLAN, _state(), review=fake)
sent = fake.calls[0]["prompt"]
assert "ship a thing" in sent
# --------------------------------------------------------------------------- #
# Fail-safe (UNTRUSTED output, never auto-approve)
# --------------------------------------------------------------------------- #
def test_review_call_raising_fails_safe() -> None:
"""A review call that raises -> REQUEST_CHANGES, never an auto-approve."""
fake = _FakeReview(raises=RuntimeError("orchestrator exploded"))
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
def test_garbage_output_fails_safe() -> None:
"""Unparseable / ambiguous reviewer text -> REQUEST_CHANGES."""
fake = _FakeReview("lorem ipsum dolor sit amet, nothing verdict-like here")
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
def test_empty_output_fails_safe() -> None:
"""Empty reviewer output -> REQUEST_CHANGES (fail closed)."""
fake = _FakeReview("")
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
def test_non_string_output_fails_safe() -> None:
"""A non-string (e.g. None) reviewer output never auto-approves."""
fake = _FakeReview(None)
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
def test_conflicting_tokens_fail_closed() -> None:
"""When both APPROVE and REQUEST CHANGES appear, fail closed (changes wins)."""
fake = _FakeReview("APPROVE in spirit but REQUEST CHANGES on phase 2.")
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
def test_no_plan_no_prompt_fails_safe() -> None:
"""No usable plan/state/prompt to review -> REQUEST_CHANGES, no review call."""
fake = _FakeReview("APPROVE")
verdict = review_plan(plan="not-a-dict", state=None, review=fake)
assert verdict is ReviewVerdict.REQUEST_CHANGES
assert fake.calls == []
# --------------------------------------------------------------------------- #
# Default routing to GPT-4.1 cross_reviewer (no network: monkeypatched)
# --------------------------------------------------------------------------- #
def test_orchestrator_root_resolves_to_run_py_parent() -> None:
"""The default reviewer resolves the orchestrator root holding run.py."""
root = resolve_orchestrator_root()
# run.py lives next to the resolved root.
assert resolve_run_py() == str(root / "run.py")
def test_default_reviewer_is_bound() -> None:
"""The module default reviewer is the run.py subprocess invoker."""
assert callable(default_plan_reviewer)
def test_default_path_invokes_run_py(monkeypatch: pytest.MonkeyPatch) -> None:
"""The default reviewer shells out to ``python3 <run_py> "<prompt>"``."""
captured: dict[str, Any] = {}
class _Completed:
returncode = 0
stdout = "VERDICT: APPROVE\nlgtm"
stderr = ""
def _fake_run(args: list[str], **kw: Any) -> _Completed:
captured["args"] = args
return _Completed()
monkeypatch.setattr(subprocess, "run", _fake_run)
# Point run.py resolution at a path that exists so the existence check passes.
monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", __file__)
invoker = make_run_py_invoker()
out = invoker("review this plan")
assert "APPROVE" in out
assert captured["args"][0] == "python3"
assert captured["args"][1] == __file__
assert captured["args"][2] == "review this plan"
def test_default_path_nonzero_exit_propagates_to_fail_safe(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A non-zero run.py exit makes review_plan fail safe to REQUEST_CHANGES."""
class _Completed:
returncode = 1
stdout = ""
stderr = "boom"
monkeypatch.setattr(subprocess, "run", lambda *a, **k: _Completed())
monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", __file__)
verdict = review_plan(_PLAN, _state()) # uses the default reviewer
assert verdict is ReviewVerdict.REQUEST_CHANGES
def test_default_path_missing_run_py_fails_safe(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A missing run.py makes the default review fail safe, not auto-approve."""
monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", "/nonexistent/path/to/run.py")
verdict = review_plan(_PLAN, _state())
assert verdict is ReviewVerdict.REQUEST_CHANGES