Remediates the Phase-0 adversarial BLOCKs: - Durable ci_pending_provider (_enumerate_ci_pending) walks the LangGraph SQLite checkpointer to enumerate threads suspended at VERIFY awaiting CI; re-derives across restart. Excludes human-clarify gates + advanced threads. - run-team serve wires ci_pending_provider + ci_poller + ci_timeout ONLY on a configured box; inert path unchanged. Closes the 'VERIFY suspended forever' defect: tick()->_ci_watch resumes on terminal CI or timeout-parks. - CI resume routes through the single-flight, turn-guarded ResumeWorker. - FIXes: run-locator skips cancelled/stale runs on rapid re-dispatch; inert-mode wording matches behavior; added node-level fail-closed + spurious-resume tests. - end-to-end async-resume proof (test_p3_async_resume.py, real checkpointer). Suite: 1270 passed, ruff clean. Branch only; not merged/deployed.
330 lines
12 KiB
Python
330 lines
12 KiB
Python
"""Unit tests for agent_team.dispatcher — the trusted apply-path transport (§4.3).
|
|
|
|
Fully hermetic: the branch-push and workflow-dispatch side effects are injected
|
|
fakes, so no git, no ``gh``, and no network are exercised. The tests pin the
|
|
pure input-assembly + validation contract and the push-before-dispatch order.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import base64
|
|
|
|
import pytest
|
|
|
|
from agent_team.dispatcher import (
|
|
MAX_DIFF_BYTES,
|
|
DispatcherError,
|
|
DispatchInputs,
|
|
DispatchResult,
|
|
build_dispatch_inputs,
|
|
dispatch_apply_verify,
|
|
head_branch_for,
|
|
run_name_for,
|
|
select_run_id,
|
|
)
|
|
from agent_team.state_store import compute_content_hash
|
|
|
|
TASK = "0a1b2c3d4e5f6071"
|
|
DIFF = "diff --git a/README.md b/README.md\n--- a/README.md\n+++ b/README.md\n@@ -1 +1,2 @@\n title\n+added line\n"
|
|
SCOPE = "README.md\ndocs/**"
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# build_dispatch_inputs (pure)
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def test_build_dispatch_inputs_binds_hash_and_b64() -> None:
|
|
di = build_dispatch_inputs(task_id=TASK, diff_text=DIFF, declared_scope=SCOPE)
|
|
# expected_diff_hash is the plain sha256 the workflow's sha256sum reproduces.
|
|
assert di.expected_diff_hash == compute_content_hash(DIFF.encode("utf-8"))
|
|
# diff_b64 round-trips back to the exact diff bytes.
|
|
assert base64.b64decode(di.diff_b64).decode("utf-8") == DIFF
|
|
assert di.head_branch == f"agent-team/apply/{TASK}"
|
|
assert di.diff_artifact_name == f"agent-team-diff-{TASK}"
|
|
assert di.declared_scope == SCOPE
|
|
|
|
|
|
def test_as_inputs_keys_match_the_six_workflow_inputs() -> None:
|
|
di = build_dispatch_inputs(task_id=TASK, diff_text=DIFF, declared_scope=SCOPE)
|
|
assert set(di.as_inputs()) == {
|
|
"task_id",
|
|
"diff_artifact_name",
|
|
"expected_diff_hash",
|
|
"declared_scope",
|
|
"diff_b64",
|
|
"head_branch",
|
|
}
|
|
|
|
|
|
@pytest.mark.parametrize("bad_diff", ["", " ", "\n\n"])
|
|
def test_empty_diff_rejected(bad_diff: str) -> None:
|
|
with pytest.raises(DispatcherError):
|
|
build_dispatch_inputs(task_id=TASK, diff_text=bad_diff, declared_scope=SCOPE)
|
|
|
|
|
|
def test_oversized_diff_rejected() -> None:
|
|
# The diff rides a base64 workflow_dispatch input (GitHub ~64 KB cap); a diff
|
|
# over MAX_DIFF_BYTES must fail closed in the dispatcher, not be dispatched.
|
|
big = "diff --git a/x b/x\n" + "+" + ("x" * (MAX_DIFF_BYTES + 1)) + "\n"
|
|
with pytest.raises(DispatcherError):
|
|
build_dispatch_inputs(task_id=TASK, diff_text=big, declared_scope=SCOPE)
|
|
|
|
|
|
@pytest.mark.parametrize("bad_scope", ["", " "])
|
|
def test_empty_scope_rejected(bad_scope: str) -> None:
|
|
# An empty declared scope would let a diff touch ANY path — fail closed.
|
|
with pytest.raises(DispatcherError):
|
|
build_dispatch_inputs(task_id=TASK, diff_text=DIFF, declared_scope=bad_scope)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"bad_task", ["", "has space", "semi;colon", "../escape", "a/b", "x" * 201]
|
|
)
|
|
def test_unsafe_task_id_rejected(bad_task: str) -> None:
|
|
with pytest.raises(DispatcherError):
|
|
head_branch_for(bad_task)
|
|
with pytest.raises(DispatcherError):
|
|
build_dispatch_inputs(task_id=bad_task, diff_text=DIFF, declared_scope=SCOPE)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# dispatch_apply_verify (injected seams)
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
class _Recorder:
|
|
def __init__(self) -> None:
|
|
self.calls: list[dict] = []
|
|
|
|
|
|
def test_dispatch_pushes_then_fires_with_correct_inputs() -> None:
|
|
order: list[str] = []
|
|
pushed = _Recorder()
|
|
fired = _Recorder()
|
|
|
|
def pusher(*, owner, repo, base, head_branch, diff_text):
|
|
order.append("push")
|
|
pushed.calls.append(
|
|
{
|
|
"owner": owner,
|
|
"repo": repo,
|
|
"base": base,
|
|
"head": head_branch,
|
|
"diff": diff_text,
|
|
}
|
|
)
|
|
|
|
def dispatcher(*, owner, repo, inputs, ref):
|
|
order.append("dispatch")
|
|
fired.calls.append({"owner": owner, "repo": repo, "inputs": inputs, "ref": ref})
|
|
|
|
located = _Recorder()
|
|
|
|
def locator(*, owner, repo, task_id, since_iso):
|
|
order.append("locate")
|
|
located.calls.append(
|
|
{"owner": owner, "repo": repo, "task_id": task_id, "since_iso": since_iso}
|
|
)
|
|
return "27990718108"
|
|
|
|
result = dispatch_apply_verify(
|
|
owner="Sea-Haven-Industries",
|
|
repo="orchestrator",
|
|
task_id=TASK,
|
|
diff_text=DIFF,
|
|
declared_scope=SCOPE,
|
|
pusher=pusher,
|
|
dispatcher=dispatcher,
|
|
locator=locator,
|
|
)
|
|
|
|
assert isinstance(result, DispatchResult)
|
|
assert isinstance(result.inputs, DispatchInputs)
|
|
# run_id is captured from the locator and surfaced for the verifier.
|
|
assert result.run_id == "27990718108"
|
|
assert result.correlation_tag == TASK
|
|
assert result.dispatched_at # stamped, non-empty
|
|
# Push BEFORE dispatch BEFORE locate (the run can only be located after it is
|
|
# triggered, and the branch must exist before the run reaches the PR step).
|
|
assert order == ["push", "dispatch", "locate"]
|
|
assert pushed.calls[0]["head"] == f"agent-team/apply/{TASK}"
|
|
assert pushed.calls[0]["diff"] == DIFF
|
|
# The dispatch carries all six inputs, including the b64 diff + head branch.
|
|
inputs = fired.calls[0]["inputs"]
|
|
assert inputs["head_branch"] == f"agent-team/apply/{TASK}"
|
|
assert base64.b64decode(inputs["diff_b64"]).decode("utf-8") == DIFF
|
|
assert inputs["expected_diff_hash"] == compute_content_hash(DIFF.encode("utf-8"))
|
|
assert fired.calls[0]["ref"] == "main"
|
|
# The locator is keyed by THIS task and the dispatched-at watermark.
|
|
assert located.calls[0]["task_id"] == TASK
|
|
assert located.calls[0]["since_iso"] == result.dispatched_at
|
|
|
|
|
|
def test_dispatch_returns_none_run_id_when_locator_cannot_resolve() -> None:
|
|
# A fired-but-unlocatable run fails closed (None run_id); never raises here.
|
|
result = dispatch_apply_verify(
|
|
owner="o",
|
|
repo="r",
|
|
task_id=TASK,
|
|
diff_text=DIFF,
|
|
declared_scope=SCOPE,
|
|
pusher=lambda **_k: None,
|
|
dispatcher=lambda **_k: None,
|
|
locator=lambda **_k: None,
|
|
)
|
|
assert isinstance(result, DispatchResult)
|
|
assert result.run_id is None
|
|
assert result.dispatched_at # still stamped for the CI-watch timeout
|
|
|
|
|
|
def test_run_name_for_matches_workflow_run_name_convention() -> None:
|
|
# Mirrors run-name: "agent-team-apply ${{ inputs.task_id }}" in the workflow.
|
|
assert run_name_for(TASK) == f"agent-team-apply {TASK}"
|
|
|
|
|
|
def test_dispatch_does_not_fire_if_push_fails() -> None:
|
|
fired = _Recorder()
|
|
|
|
def failing_pusher(**_kw):
|
|
raise RuntimeError("push failed")
|
|
|
|
def dispatcher(**kw):
|
|
fired.calls.append(kw)
|
|
|
|
with pytest.raises(RuntimeError):
|
|
dispatch_apply_verify(
|
|
owner="o",
|
|
repo="r",
|
|
task_id=TASK,
|
|
diff_text=DIFF,
|
|
declared_scope=SCOPE,
|
|
pusher=failing_pusher,
|
|
dispatcher=dispatcher,
|
|
)
|
|
# A failed push must NOT dispatch a run (no orphan run against a missing head).
|
|
assert fired.calls == []
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"owner,repo", [("", "r"), ("o", ""), ("bad owner", "r"), ("o", "r/x")]
|
|
)
|
|
def test_unsafe_owner_repo_rejected(owner: str, repo: str) -> None:
|
|
with pytest.raises(DispatcherError):
|
|
dispatch_apply_verify(
|
|
owner=owner,
|
|
repo=repo,
|
|
task_id=TASK,
|
|
diff_text=DIFF,
|
|
declared_scope=SCOPE,
|
|
pusher=lambda **_k: None,
|
|
dispatcher=lambda **_k: None,
|
|
)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# select_run_id (pure; anti-stale on rapid re-dispatch of the SAME task_id)
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def _row(db_id, *, created, status="completed", conclusion=None):
|
|
"""A minimal ``gh run list`` row for the apply/verify run-name of TASK."""
|
|
return {
|
|
"databaseId": db_id,
|
|
"name": run_name_for(TASK),
|
|
"createdAt": created,
|
|
"status": status,
|
|
"conclusion": conclusion,
|
|
}
|
|
|
|
|
|
def test_select_run_id_skips_cancelled_prior_run_on_re_dispatch() -> None:
|
|
# Rapid re-dispatch of the SAME task_id: the concurrency group cancelled the
|
|
# OLDER run, and a NEWER run is now in progress. We must bind to the newer,
|
|
# active run — never the older cancelled one (it carries the prior verdict).
|
|
older_cancelled = _row(
|
|
100, created="2026-06-23T10:00:00Z", status="completed", conclusion="cancelled"
|
|
)
|
|
newer_active = _row(
|
|
200, created="2026-06-23T10:05:00Z", status="in_progress", conclusion=None
|
|
)
|
|
runs = [newer_active, older_cancelled]
|
|
chosen = select_run_id(runs, task_id=TASK, floor_iso="2026-06-23T09:58:00Z")
|
|
assert chosen == "200"
|
|
|
|
|
|
def test_select_run_id_skips_cancelled_even_when_it_is_newest() -> None:
|
|
# Defensive: a cancelled run is NEVER selected even if its createdAt is the
|
|
# greatest — it is the superseded run, not ours.
|
|
active = _row(300, created="2026-06-23T10:00:00Z", status="queued", conclusion=None)
|
|
newest_cancelled = _row(
|
|
400, created="2026-06-23T10:10:00Z", status="completed", conclusion="cancelled"
|
|
)
|
|
runs = [active, newest_cancelled]
|
|
chosen = select_run_id(runs, task_id=TASK, floor_iso="2026-06-23T09:58:00Z")
|
|
assert chosen == "300"
|
|
|
|
|
|
def test_select_run_id_prefers_newest_active_over_older_completed() -> None:
|
|
# An older legitimately-completed run plus a newer active run -> the active,
|
|
# newest run wins (the freshly-triggered one with no conclusion yet).
|
|
older_done = _row(
|
|
500, created="2026-06-23T10:00:00Z", status="completed", conclusion="success"
|
|
)
|
|
newer_active = _row(
|
|
600, created="2026-06-23T10:05:00Z", status="in_progress", conclusion=None
|
|
)
|
|
chosen = select_run_id(
|
|
[older_done, newer_active], task_id=TASK, floor_iso="2026-06-23T09:58:00Z"
|
|
)
|
|
assert chosen == "600"
|
|
|
|
|
|
def test_select_run_id_falls_back_to_newest_non_cancelled_when_none_active() -> None:
|
|
# No active runs (e.g. a fast run already concluded by the time we poll):
|
|
# fall back to the newest NON-cancelled run overall.
|
|
older = _row(
|
|
700, created="2026-06-23T10:00:00Z", status="completed", conclusion="success"
|
|
)
|
|
newer = _row(
|
|
800, created="2026-06-23T10:05:00Z", status="completed", conclusion="failure"
|
|
)
|
|
cancelled = _row(
|
|
900, created="2026-06-23T10:09:00Z", status="completed", conclusion="cancelled"
|
|
)
|
|
chosen = select_run_id(
|
|
[older, newer, cancelled], task_id=TASK, floor_iso="2026-06-23T09:58:00Z"
|
|
)
|
|
assert chosen == "800"
|
|
|
|
|
|
def test_select_run_id_respects_created_floor_and_run_name() -> None:
|
|
# Below-floor runs and other-task runs are not matched.
|
|
below_floor = _row(
|
|
1000, created="2026-06-23T09:00:00Z", status="in_progress", conclusion=None
|
|
)
|
|
other_task = {
|
|
"databaseId": 1100,
|
|
"name": "agent-team-apply other-task",
|
|
"createdAt": "2026-06-23T10:00:00Z",
|
|
"status": "in_progress",
|
|
"conclusion": None,
|
|
}
|
|
assert (
|
|
select_run_id(
|
|
[below_floor, other_task], task_id=TASK, floor_iso="2026-06-23T09:58:00Z"
|
|
)
|
|
is None
|
|
)
|
|
|
|
|
|
def test_select_run_id_returns_none_when_only_cancelled_matches() -> None:
|
|
# If the only matching run is cancelled, there is nothing to bind to -> None
|
|
# (the caller fails closed: no run_id -> verify BLOCKs/parks).
|
|
only_cancelled = _row(
|
|
1200, created="2026-06-23T10:00:00Z", status="completed", conclusion="cancelled"
|
|
)
|
|
assert (
|
|
select_run_id([only_cancelled], task_id=TASK, floor_iso="2026-06-23T09:58:00Z")
|
|
is None
|
|
)
|