230 lines
8.7 KiB
Python
230 lines
8.7 KiB
Python
|
|
"""Unit tests for agent_team.nodes.review_loop_llm (§3.3, §7.1 P2).
|
||
|
|
|
||
|
|
The real GPT-4.1 cross-family review binding is exercised with a FAKE review
|
||
|
|
callable that returns canned verdict text — no network, no subprocess. The
|
||
|
|
load-bearing properties under test:
|
||
|
|
|
||
|
|
* **No orchestrator import at module load.** Importing this module must not pull
|
||
|
|
in the orchestrator package (``models`` / ``graph``); the default reviewer
|
||
|
|
shells out / imports lazily.
|
||
|
|
* **Clear approve -> APPROVE.** An injected fake returning an explicit APPROVE
|
||
|
|
verdict maps to the node-contract ``ReviewVerdict.APPROVE`` (proceed).
|
||
|
|
* **Changes requested -> REQUEST_CHANGES.** The loop-back / escalate verdict.
|
||
|
|
* **Fail SAFE.** A review call that raises, or returns garbage / empty / a
|
||
|
|
non-string, maps to ``REQUEST_CHANGES`` — never an auto-approve.
|
||
|
|
* **Routing facts.** The default reviewer resolves the orchestrator root at
|
||
|
|
``parents[3]`` and a ``run.py`` next to it, and is bound as the default.
|
||
|
|
"""
|
||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import subprocess
|
||
|
|
import sys
|
||
|
|
from typing import Any
|
||
|
|
|
||
|
|
import pytest
|
||
|
|
|
||
|
|
from agent_team.nodes.review_loop import ReviewVerdict
|
||
|
|
from agent_team.nodes.review_loop_llm import (
|
||
|
|
default_plan_reviewer,
|
||
|
|
make_run_py_invoker,
|
||
|
|
resolve_orchestrator_root,
|
||
|
|
resolve_run_py,
|
||
|
|
review_plan,
|
||
|
|
)
|
||
|
|
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
# Fakes / helpers
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
|
||
|
|
|
||
|
|
class _FakeReview:
|
||
|
|
"""A fake plan reviewer that returns canned text and records its calls.
|
||
|
|
|
||
|
|
``reply`` is the verdict text returned every call. ``raises`` (if set) is
|
||
|
|
raised instead, to simulate a failed review call.
|
||
|
|
"""
|
||
|
|
|
||
|
|
def __init__(self, reply: Any = "", *, raises: BaseException | None = None) -> None:
|
||
|
|
self._reply = reply
|
||
|
|
self._raises = raises
|
||
|
|
self.calls: list[dict[str, Any]] = []
|
||
|
|
|
||
|
|
def __call__(self, prompt: str, **kw: Any) -> Any:
|
||
|
|
self.calls.append({"prompt": prompt, "kw": kw})
|
||
|
|
if self._raises is not None:
|
||
|
|
raise self._raises
|
||
|
|
return self._reply
|
||
|
|
|
||
|
|
|
||
|
|
_PLAN = {"task": "ship a thing", "phases": [{"name": "P1"}, {"name": "P2"}]}
|
||
|
|
|
||
|
|
|
||
|
|
def _state() -> dict[str, Any]:
|
||
|
|
return {"plan": _PLAN, "review_verdicts": []}
|
||
|
|
|
||
|
|
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
# Module import hygiene
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
|
||
|
|
|
||
|
|
def test_module_imports_without_orchestrator() -> None:
|
||
|
|
"""Importing the module must not import the orchestrator package."""
|
||
|
|
# The module is already imported at top, but assert the orchestrator stack
|
||
|
|
# did not get pulled in as a side effect of importing it.
|
||
|
|
assert "models" not in sys.modules
|
||
|
|
assert "graph" not in sys.modules
|
||
|
|
|
||
|
|
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
# Verdict mapping
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
|
||
|
|
|
||
|
|
def test_clear_approve_maps_to_approve() -> None:
|
||
|
|
"""An injected fake returning a clear APPROVE verdict -> ReviewVerdict.APPROVE."""
|
||
|
|
fake = _FakeReview("VERDICT: APPROVE\nThe plan is sound and ready to build.")
|
||
|
|
verdict = review_plan(_PLAN, _state(), review=fake)
|
||
|
|
assert verdict is ReviewVerdict.APPROVE
|
||
|
|
|
||
|
|
|
||
|
|
def test_changes_requested_maps_to_request_changes() -> None:
|
||
|
|
"""An injected fake returning changes-requested -> ReviewVerdict.REQUEST_CHANGES."""
|
||
|
|
fake = _FakeReview("VERDICT: REQUEST CHANGES\nPhase ordering is wrong.")
|
||
|
|
verdict = review_plan(_PLAN, _state(), review=fake)
|
||
|
|
assert verdict is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_block_token_maps_to_request_changes() -> None:
|
||
|
|
"""A BLOCK verdict (sh-plan-review vocabulary) -> REQUEST_CHANGES."""
|
||
|
|
fake = _FakeReview("BLOCK: missing rollback phase.")
|
||
|
|
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_prompt_only_calling_convention() -> None:
|
||
|
|
"""review_plan(prompt=...) reviews an already-composed prompt (the seam shape)."""
|
||
|
|
fake = _FakeReview("APPROVE")
|
||
|
|
verdict = review_plan(prompt="pre-composed review task", review=fake)
|
||
|
|
assert verdict is ReviewVerdict.APPROVE
|
||
|
|
assert fake.calls[0]["prompt"] == "pre-composed review task"
|
||
|
|
|
||
|
|
|
||
|
|
def test_prompt_is_composed_from_plan_when_not_supplied() -> None:
|
||
|
|
"""With no prompt, the plan text is embedded in the composed review task."""
|
||
|
|
fake = _FakeReview("APPROVE")
|
||
|
|
review_plan(_PLAN, _state(), review=fake)
|
||
|
|
sent = fake.calls[0]["prompt"]
|
||
|
|
assert "ship a thing" in sent
|
||
|
|
|
||
|
|
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
# Fail-safe (UNTRUSTED output, never auto-approve)
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
|
||
|
|
|
||
|
|
def test_review_call_raising_fails_safe() -> None:
|
||
|
|
"""A review call that raises -> REQUEST_CHANGES, never an auto-approve."""
|
||
|
|
fake = _FakeReview(raises=RuntimeError("orchestrator exploded"))
|
||
|
|
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_garbage_output_fails_safe() -> None:
|
||
|
|
"""Unparseable / ambiguous reviewer text -> REQUEST_CHANGES."""
|
||
|
|
fake = _FakeReview("lorem ipsum dolor sit amet, nothing verdict-like here")
|
||
|
|
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_empty_output_fails_safe() -> None:
|
||
|
|
"""Empty reviewer output -> REQUEST_CHANGES (fail closed)."""
|
||
|
|
fake = _FakeReview("")
|
||
|
|
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_non_string_output_fails_safe() -> None:
|
||
|
|
"""A non-string (e.g. None) reviewer output never auto-approves."""
|
||
|
|
fake = _FakeReview(None)
|
||
|
|
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_conflicting_tokens_fail_closed() -> None:
|
||
|
|
"""When both APPROVE and REQUEST CHANGES appear, fail closed (changes wins)."""
|
||
|
|
fake = _FakeReview("APPROVE in spirit but REQUEST CHANGES on phase 2.")
|
||
|
|
assert review_plan(_PLAN, _state(), review=fake) is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_no_plan_no_prompt_fails_safe() -> None:
|
||
|
|
"""No usable plan/state/prompt to review -> REQUEST_CHANGES, no review call."""
|
||
|
|
fake = _FakeReview("APPROVE")
|
||
|
|
verdict = review_plan(plan="not-a-dict", state=None, review=fake)
|
||
|
|
assert verdict is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
assert fake.calls == []
|
||
|
|
|
||
|
|
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
# Default routing to GPT-4.1 cross_reviewer (no network: monkeypatched)
|
||
|
|
# --------------------------------------------------------------------------- #
|
||
|
|
|
||
|
|
|
||
|
|
def test_orchestrator_root_resolves_to_run_py_parent() -> None:
|
||
|
|
"""The default reviewer resolves the orchestrator root holding run.py."""
|
||
|
|
root = resolve_orchestrator_root()
|
||
|
|
# run.py lives next to the resolved root.
|
||
|
|
assert resolve_run_py() == str(root / "run.py")
|
||
|
|
|
||
|
|
|
||
|
|
def test_default_reviewer_is_bound() -> None:
|
||
|
|
"""The module default reviewer is the run.py subprocess invoker."""
|
||
|
|
assert callable(default_plan_reviewer)
|
||
|
|
|
||
|
|
|
||
|
|
def test_default_path_invokes_run_py(monkeypatch: pytest.MonkeyPatch) -> None:
|
||
|
|
"""The default reviewer shells out to ``python3 <run_py> "<prompt>"``."""
|
||
|
|
captured: dict[str, Any] = {}
|
||
|
|
|
||
|
|
class _Completed:
|
||
|
|
returncode = 0
|
||
|
|
stdout = "VERDICT: APPROVE\nlgtm"
|
||
|
|
stderr = ""
|
||
|
|
|
||
|
|
def _fake_run(args: list[str], **kw: Any) -> _Completed:
|
||
|
|
captured["args"] = args
|
||
|
|
return _Completed()
|
||
|
|
|
||
|
|
monkeypatch.setattr(subprocess, "run", _fake_run)
|
||
|
|
# Point run.py resolution at a path that exists so the existence check passes.
|
||
|
|
monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", __file__)
|
||
|
|
|
||
|
|
invoker = make_run_py_invoker()
|
||
|
|
out = invoker("review this plan")
|
||
|
|
assert "APPROVE" in out
|
||
|
|
assert captured["args"][0] == "python3"
|
||
|
|
assert captured["args"][1] == __file__
|
||
|
|
assert captured["args"][2] == "review this plan"
|
||
|
|
|
||
|
|
|
||
|
|
def test_default_path_nonzero_exit_propagates_to_fail_safe(
|
||
|
|
monkeypatch: pytest.MonkeyPatch,
|
||
|
|
) -> None:
|
||
|
|
"""A non-zero run.py exit makes review_plan fail safe to REQUEST_CHANGES."""
|
||
|
|
|
||
|
|
class _Completed:
|
||
|
|
returncode = 1
|
||
|
|
stdout = ""
|
||
|
|
stderr = "boom"
|
||
|
|
|
||
|
|
monkeypatch.setattr(subprocess, "run", lambda *a, **k: _Completed())
|
||
|
|
monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", __file__)
|
||
|
|
|
||
|
|
verdict = review_plan(_PLAN, _state()) # uses the default reviewer
|
||
|
|
assert verdict is ReviewVerdict.REQUEST_CHANGES
|
||
|
|
|
||
|
|
|
||
|
|
def test_default_path_missing_run_py_fails_safe(
|
||
|
|
monkeypatch: pytest.MonkeyPatch,
|
||
|
|
) -> None:
|
||
|
|
"""A missing run.py makes the default review fail safe, not auto-approve."""
|
||
|
|
monkeypatch.setenv("AGENT_TEAM_ORCHESTRATOR_RUN_PY", "/nonexistent/path/to/run.py")
|
||
|
|
verdict = review_plan(_PLAN, _state())
|
||
|
|
assert verdict is ReviewVerdict.REQUEST_CHANGES
|