This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/tests/test_dispatcher.py
Adam Moussa 1f8c7e1ee3 fix(agent-team): close P3 async-resume BLOCKs (durable CI-watcher wiring)
Remediates the Phase-0 adversarial BLOCKs:
- Durable ci_pending_provider (_enumerate_ci_pending) walks the LangGraph
  SQLite checkpointer to enumerate threads suspended at VERIFY awaiting CI;
  re-derives across restart. Excludes human-clarify gates + advanced threads.
- run-team serve wires ci_pending_provider + ci_poller + ci_timeout ONLY on a
  configured box; inert path unchanged. Closes the 'VERIFY suspended forever'
  defect: tick()->_ci_watch resumes on terminal CI or timeout-parks.
- CI resume routes through the single-flight, turn-guarded ResumeWorker.
- FIXes: run-locator skips cancelled/stale runs on rapid re-dispatch; inert-mode
  wording matches behavior; added node-level fail-closed + spurious-resume tests.
- end-to-end async-resume proof (test_p3_async_resume.py, real checkpointer).

Suite: 1270 passed, ruff clean. Branch only; not merged/deployed.
2026-06-23 19:52:04 -04:00

330 lines
12 KiB
Python

"""Unit tests for agent_team.dispatcher — the trusted apply-path transport (§4.3).
Fully hermetic: the branch-push and workflow-dispatch side effects are injected
fakes, so no git, no ``gh``, and no network are exercised. The tests pin the
pure input-assembly + validation contract and the push-before-dispatch order.
"""
from __future__ import annotations
import base64
import pytest
from agent_team.dispatcher import (
MAX_DIFF_BYTES,
DispatcherError,
DispatchInputs,
DispatchResult,
build_dispatch_inputs,
dispatch_apply_verify,
head_branch_for,
run_name_for,
select_run_id,
)
from agent_team.state_store import compute_content_hash
TASK = "0a1b2c3d4e5f6071"
DIFF = "diff --git a/README.md b/README.md\n--- a/README.md\n+++ b/README.md\n@@ -1 +1,2 @@\n title\n+added line\n"
SCOPE = "README.md\ndocs/**"
# --------------------------------------------------------------------------- #
# build_dispatch_inputs (pure)
# --------------------------------------------------------------------------- #
def test_build_dispatch_inputs_binds_hash_and_b64() -> None:
di = build_dispatch_inputs(task_id=TASK, diff_text=DIFF, declared_scope=SCOPE)
# expected_diff_hash is the plain sha256 the workflow's sha256sum reproduces.
assert di.expected_diff_hash == compute_content_hash(DIFF.encode("utf-8"))
# diff_b64 round-trips back to the exact diff bytes.
assert base64.b64decode(di.diff_b64).decode("utf-8") == DIFF
assert di.head_branch == f"agent-team/apply/{TASK}"
assert di.diff_artifact_name == f"agent-team-diff-{TASK}"
assert di.declared_scope == SCOPE
def test_as_inputs_keys_match_the_six_workflow_inputs() -> None:
di = build_dispatch_inputs(task_id=TASK, diff_text=DIFF, declared_scope=SCOPE)
assert set(di.as_inputs()) == {
"task_id",
"diff_artifact_name",
"expected_diff_hash",
"declared_scope",
"diff_b64",
"head_branch",
}
@pytest.mark.parametrize("bad_diff", ["", " ", "\n\n"])
def test_empty_diff_rejected(bad_diff: str) -> None:
with pytest.raises(DispatcherError):
build_dispatch_inputs(task_id=TASK, diff_text=bad_diff, declared_scope=SCOPE)
def test_oversized_diff_rejected() -> None:
# The diff rides a base64 workflow_dispatch input (GitHub ~64 KB cap); a diff
# over MAX_DIFF_BYTES must fail closed in the dispatcher, not be dispatched.
big = "diff --git a/x b/x\n" + "+" + ("x" * (MAX_DIFF_BYTES + 1)) + "\n"
with pytest.raises(DispatcherError):
build_dispatch_inputs(task_id=TASK, diff_text=big, declared_scope=SCOPE)
@pytest.mark.parametrize("bad_scope", ["", " "])
def test_empty_scope_rejected(bad_scope: str) -> None:
# An empty declared scope would let a diff touch ANY path — fail closed.
with pytest.raises(DispatcherError):
build_dispatch_inputs(task_id=TASK, diff_text=DIFF, declared_scope=bad_scope)
@pytest.mark.parametrize(
"bad_task", ["", "has space", "semi;colon", "../escape", "a/b", "x" * 201]
)
def test_unsafe_task_id_rejected(bad_task: str) -> None:
with pytest.raises(DispatcherError):
head_branch_for(bad_task)
with pytest.raises(DispatcherError):
build_dispatch_inputs(task_id=bad_task, diff_text=DIFF, declared_scope=SCOPE)
# --------------------------------------------------------------------------- #
# dispatch_apply_verify (injected seams)
# --------------------------------------------------------------------------- #
class _Recorder:
def __init__(self) -> None:
self.calls: list[dict] = []
def test_dispatch_pushes_then_fires_with_correct_inputs() -> None:
order: list[str] = []
pushed = _Recorder()
fired = _Recorder()
def pusher(*, owner, repo, base, head_branch, diff_text):
order.append("push")
pushed.calls.append(
{
"owner": owner,
"repo": repo,
"base": base,
"head": head_branch,
"diff": diff_text,
}
)
def dispatcher(*, owner, repo, inputs, ref):
order.append("dispatch")
fired.calls.append({"owner": owner, "repo": repo, "inputs": inputs, "ref": ref})
located = _Recorder()
def locator(*, owner, repo, task_id, since_iso):
order.append("locate")
located.calls.append(
{"owner": owner, "repo": repo, "task_id": task_id, "since_iso": since_iso}
)
return "27990718108"
result = dispatch_apply_verify(
owner="Sea-Haven-Industries",
repo="orchestrator",
task_id=TASK,
diff_text=DIFF,
declared_scope=SCOPE,
pusher=pusher,
dispatcher=dispatcher,
locator=locator,
)
assert isinstance(result, DispatchResult)
assert isinstance(result.inputs, DispatchInputs)
# run_id is captured from the locator and surfaced for the verifier.
assert result.run_id == "27990718108"
assert result.correlation_tag == TASK
assert result.dispatched_at # stamped, non-empty
# Push BEFORE dispatch BEFORE locate (the run can only be located after it is
# triggered, and the branch must exist before the run reaches the PR step).
assert order == ["push", "dispatch", "locate"]
assert pushed.calls[0]["head"] == f"agent-team/apply/{TASK}"
assert pushed.calls[0]["diff"] == DIFF
# The dispatch carries all six inputs, including the b64 diff + head branch.
inputs = fired.calls[0]["inputs"]
assert inputs["head_branch"] == f"agent-team/apply/{TASK}"
assert base64.b64decode(inputs["diff_b64"]).decode("utf-8") == DIFF
assert inputs["expected_diff_hash"] == compute_content_hash(DIFF.encode("utf-8"))
assert fired.calls[0]["ref"] == "main"
# The locator is keyed by THIS task and the dispatched-at watermark.
assert located.calls[0]["task_id"] == TASK
assert located.calls[0]["since_iso"] == result.dispatched_at
def test_dispatch_returns_none_run_id_when_locator_cannot_resolve() -> None:
# A fired-but-unlocatable run fails closed (None run_id); never raises here.
result = dispatch_apply_verify(
owner="o",
repo="r",
task_id=TASK,
diff_text=DIFF,
declared_scope=SCOPE,
pusher=lambda **_k: None,
dispatcher=lambda **_k: None,
locator=lambda **_k: None,
)
assert isinstance(result, DispatchResult)
assert result.run_id is None
assert result.dispatched_at # still stamped for the CI-watch timeout
def test_run_name_for_matches_workflow_run_name_convention() -> None:
# Mirrors run-name: "agent-team-apply ${{ inputs.task_id }}" in the workflow.
assert run_name_for(TASK) == f"agent-team-apply {TASK}"
def test_dispatch_does_not_fire_if_push_fails() -> None:
fired = _Recorder()
def failing_pusher(**_kw):
raise RuntimeError("push failed")
def dispatcher(**kw):
fired.calls.append(kw)
with pytest.raises(RuntimeError):
dispatch_apply_verify(
owner="o",
repo="r",
task_id=TASK,
diff_text=DIFF,
declared_scope=SCOPE,
pusher=failing_pusher,
dispatcher=dispatcher,
)
# A failed push must NOT dispatch a run (no orphan run against a missing head).
assert fired.calls == []
@pytest.mark.parametrize(
"owner,repo", [("", "r"), ("o", ""), ("bad owner", "r"), ("o", "r/x")]
)
def test_unsafe_owner_repo_rejected(owner: str, repo: str) -> None:
with pytest.raises(DispatcherError):
dispatch_apply_verify(
owner=owner,
repo=repo,
task_id=TASK,
diff_text=DIFF,
declared_scope=SCOPE,
pusher=lambda **_k: None,
dispatcher=lambda **_k: None,
)
# --------------------------------------------------------------------------- #
# select_run_id (pure; anti-stale on rapid re-dispatch of the SAME task_id)
# --------------------------------------------------------------------------- #
def _row(db_id, *, created, status="completed", conclusion=None):
"""A minimal ``gh run list`` row for the apply/verify run-name of TASK."""
return {
"databaseId": db_id,
"name": run_name_for(TASK),
"createdAt": created,
"status": status,
"conclusion": conclusion,
}
def test_select_run_id_skips_cancelled_prior_run_on_re_dispatch() -> None:
# Rapid re-dispatch of the SAME task_id: the concurrency group cancelled the
# OLDER run, and a NEWER run is now in progress. We must bind to the newer,
# active run — never the older cancelled one (it carries the prior verdict).
older_cancelled = _row(
100, created="2026-06-23T10:00:00Z", status="completed", conclusion="cancelled"
)
newer_active = _row(
200, created="2026-06-23T10:05:00Z", status="in_progress", conclusion=None
)
runs = [newer_active, older_cancelled]
chosen = select_run_id(runs, task_id=TASK, floor_iso="2026-06-23T09:58:00Z")
assert chosen == "200"
def test_select_run_id_skips_cancelled_even_when_it_is_newest() -> None:
# Defensive: a cancelled run is NEVER selected even if its createdAt is the
# greatest — it is the superseded run, not ours.
active = _row(300, created="2026-06-23T10:00:00Z", status="queued", conclusion=None)
newest_cancelled = _row(
400, created="2026-06-23T10:10:00Z", status="completed", conclusion="cancelled"
)
runs = [active, newest_cancelled]
chosen = select_run_id(runs, task_id=TASK, floor_iso="2026-06-23T09:58:00Z")
assert chosen == "300"
def test_select_run_id_prefers_newest_active_over_older_completed() -> None:
# An older legitimately-completed run plus a newer active run -> the active,
# newest run wins (the freshly-triggered one with no conclusion yet).
older_done = _row(
500, created="2026-06-23T10:00:00Z", status="completed", conclusion="success"
)
newer_active = _row(
600, created="2026-06-23T10:05:00Z", status="in_progress", conclusion=None
)
chosen = select_run_id(
[older_done, newer_active], task_id=TASK, floor_iso="2026-06-23T09:58:00Z"
)
assert chosen == "600"
def test_select_run_id_falls_back_to_newest_non_cancelled_when_none_active() -> None:
# No active runs (e.g. a fast run already concluded by the time we poll):
# fall back to the newest NON-cancelled run overall.
older = _row(
700, created="2026-06-23T10:00:00Z", status="completed", conclusion="success"
)
newer = _row(
800, created="2026-06-23T10:05:00Z", status="completed", conclusion="failure"
)
cancelled = _row(
900, created="2026-06-23T10:09:00Z", status="completed", conclusion="cancelled"
)
chosen = select_run_id(
[older, newer, cancelled], task_id=TASK, floor_iso="2026-06-23T09:58:00Z"
)
assert chosen == "800"
def test_select_run_id_respects_created_floor_and_run_name() -> None:
# Below-floor runs and other-task runs are not matched.
below_floor = _row(
1000, created="2026-06-23T09:00:00Z", status="in_progress", conclusion=None
)
other_task = {
"databaseId": 1100,
"name": "agent-team-apply other-task",
"createdAt": "2026-06-23T10:00:00Z",
"status": "in_progress",
"conclusion": None,
}
assert (
select_run_id(
[below_floor, other_task], task_id=TASK, floor_iso="2026-06-23T09:58:00Z"
)
is None
)
def test_select_run_id_returns_none_when_only_cancelled_matches() -> None:
# If the only matching run is cancelled, there is nothing to bind to -> None
# (the caller fails closed: no run_id -> verify BLOCKs/parks).
only_cancelled = _row(
1200, created="2026-06-23T10:00:00Z", status="completed", conclusion="cancelled"
)
assert (
select_run_id([only_cancelled], task_id=TASK, floor_iso="2026-06-23T09:58:00Z")
is None
)