"""Unit tests for agent_team.nodes.review_loop_llm (§3.3, §7.1 P2). The real GPT-4.1 cross-family review binding is exercised with a FAKE review callable that returns canned verdict text — no network, no subprocess. The load-bearing properties under test: * **No orchestrator import at module load.** Importing this module must not pull in the orchestrator package (``models`` / ``graph``); the default reviewer shells out / imports lazily. * **Clear approve -> APPROVE.** An injected fake returning an explicit APPROVE verdict maps to the node-contract ``ReviewVerdict.APPROVE`` (proceed). * **Changes requested -> REQUEST_CHANGES.** The loop-back / escalate verdict. * **Fail SAFE.** A review call that raises, or returns garbage / empty / a non-string, maps to ``REQUEST_CHANGES`` — never an auto-approve. * **Routing facts.** The default reviewer resolves the orchestrator root at ``parents[3]`` and a ``run.py`` next to it, and is bound as the default. """ from __future__ import annotations import subprocess import sys from typing import Any import pytest from agent_team.nodes.review_loop import ReviewVerdict from agent_team.nodes.review_loop_llm import ( default_plan_reviewer, make_run_py_invoker, resolve_orchestrator_root, resolve_run_py, review_plan, ) # --------------------------------------------------------------------------- # # Fakes / helpers # --------------------------------------------------------------------------- # class _FakeReview: """A fake plan reviewer that returns canned text and records its calls. ``reply`` is the verdict text returned every call. ``raises`` (if set) is raised instead, to simulate a failed review call. """ def __init__(self, reply: Any = "", *, raises: BaseException | None = None) -> None: self._reply = reply self._raises = raises self.calls: list[dict[str, Any]] = [] def __call__(self, prompt: str, **kw: Any) -> Any: self.calls.append({"prompt": prompt, "kw": kw}) if self._raises is not None: raise self._raises return self._reply _PLAN = {"task": "ship a thing", "phases": [{"name": "P1"}, {"name": "P2"}]} def _state() -> dict[str, Any]: return {"plan": _PLAN, "review_verdicts": []} # --------------------------------------------------------------------------- # # Module import hygiene # --------------------------------------------------------------------------- # def test_module_imports_without_orchestrator() -> None: """Importing the module must not import the orchestrator package.""" # The module is already imported at top, but assert the orchestrator stack # did not get pulled in as a side effect of importing it. assert "models" not in sys.modules assert "graph" not in sys.modules # --------------------------------------------------------------------------- # # Verdict mapping # --------------------------------------------------------------------------- # def test_clear_approve_maps_to_approve() -> None: """An injected fake returning a clear APPROVE verdict -> ReviewVerdict.APPROVE.""" fake = _FakeReview("VERDICT: APPROVE\nThe plan is sound and ready to build.") verdict = review_plan(_PLAN, _state(), review=fake) assert verdict is ReviewVerdict.APPROVE def test_changes_requested_maps_to_request_changes() -> None: """An injected fake returning changes-requested -> ReviewVerdict.REQUEST_CHANGES.""" fake = _FakeReview("VERDICT: REQUEST CHANGES\nPhase ordering is wrong.") verdict = review_plan(_PLAN, _state(), review=fake) assert verdict is ReviewVerdict.REQUEST_CHANGES def test_block_token_maps_to_request_changes() -> None: """A BLOCK verdict (sh-plan-review vocabulary) -> REQUEST_CHANGES.""" fake = _FakeReview("BLOCK: missing rollback phase.") assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES def test_prompt_only_calling_convention() -> None: """review_plan(prompt=...) reviews an already-composed prompt (the seam shape).""" fake = _FakeReview("APPROVE") verdict = review_plan(prompt="pre-composed review task", review=fake) assert verdict is ReviewVerdict.APPROVE assert fake.calls[0]["prompt"] == "pre-composed review task" def test_prompt_is_composed_from_plan_when_not_supplied() -> None: """With no prompt, the plan text is embedded in the composed review task.""" fake = _FakeReview("APPROVE") review_plan(_PLAN, _state(), review=fake) sent = fake.calls[0]["prompt"] assert "ship a thing" in sent # --------------------------------------------------------------------------- # # Fail-safe (UNTRUSTED output, never auto-approve) # --------------------------------------------------------------------------- # def test_review_call_raising_fails_safe() -> None: """A review call that raises -> REQUEST_CHANGES, never an auto-approve.""" fake = _FakeReview(raises=RuntimeError("orchestrator exploded")) assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES def test_garbage_output_fails_safe() -> None: """Unparseable / ambiguous reviewer text -> REQUEST_CHANGES.""" fake = _FakeReview("lorem ipsum dolor sit amet, nothing verdict-like here") assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES def test_empty_output_fails_safe() -> None: """Empty reviewer output -> REQUEST_CHANGES (fail closed).""" fake = _FakeReview("") assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES def test_non_string_output_fails_safe() -> None: """A non-string (e.g. None) reviewer output never auto-approves.""" fake = _FakeReview(None) assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES def test_conflicting_tokens_fail_closed() -> None: """When both APPROVE and REQUEST CHANGES appear, fail closed (changes wins).""" fake = _FakeReview("APPROVE in spirit but REQUEST CHANGES on phase 2.") assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES def test_no_plan_no_prompt_fails_safe() -> None: """No usable plan/state/prompt to review -> REQUEST_CHANGES, no review call.""" fake = _FakeReview("APPROVE") verdict = review_plan(plan="not-a-dict", state=None, review=fake) assert verdict is ReviewVerdict.REQUEST_CHANGES assert fake.calls == [] # --------------------------------------------------------------------------- # # Default routing to GPT-4.1 cross_reviewer (no network: monkeypatched) # --------------------------------------------------------------------------- # def test_orchestrator_root_resolves_to_run_py_parent() -> None: """The default reviewer resolves the orchestrator root holding run.py.""" root = resolve_orchestrator_root() # run.py lives next to the resolved root. assert resolve_run_py() == str(root / "run.py") def test_default_reviewer_is_bound() -> None: """The module default reviewer is the run.py subprocess invoker.""" assert callable(default_plan_reviewer) def test_default_path_invokes_run_py(monkeypatch: pytest.MonkeyPatch) -> None: """The default reviewer shells out to ``python3 ""``.""" captured: dict[str, Any] = {} class _Completed: returncode = 0 stdout = "VERDICT: APPROVE\nlgtm" stderr = "" def _fake_run(args: list[str], **kw: Any) -> _Completed: captured["args"] = args return _Completed() monkeypatch.setattr(subprocess, "run", _fake_run) # Point run.py resolution at a path that exists so the existence check passes. monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", __file__) invoker = make_run_py_invoker() out = invoker("review this plan") assert "APPROVE" in out assert captured["args"][0] == "python3" assert captured["args"][1] == __file__ assert captured["args"][2] == "review this plan" def test_default_path_nonzero_exit_propagates_to_fail_safe( monkeypatch: pytest.MonkeyPatch, ) -> None: """A non-zero run.py exit makes review_plan fail safe to REQUEST_CHANGES.""" class _Completed: returncode = 1 stdout = "" stderr = "boom" monkeypatch.setattr(subprocess, "run", lambda *a, **k: _Completed()) monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", __file__) verdict = review_plan(_PLAN, _state()) # uses the default reviewer assert verdict is ReviewVerdict.REQUEST_CHANGES def test_default_path_missing_run_py_fails_safe( monkeypatch: pytest.MonkeyPatch, ) -> None: """A missing run.py makes the default review fail safe, not auto-approve.""" monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", "/nonexistent/path/to/run.py") verdict = review_plan(_PLAN, _state()) assert verdict is ReviewVerdict.REQUEST_CHANGES