This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/agent_team/task_model.py

231 lines
9.4 KiB
Python
Raw Normal View History

"""Task-record / thread model + LangGraph graph-state schema (design §3.3).
A task is a long-lived, resumable record (a LangGraph thread). This module is
the pure model layer — no I/O — defining:
* :class:`TaskStatus` / :class:`Phase` — task lifecycle enums.
* :class:`TaskRecord` — the durable task record (§3.3 "the task record holds:
status, current phase, the full Q&A history, the plan, review verdicts, the
candidate diff, and CI results").
* :class:`PipelineState` — a ``TypedDict`` used as the LangGraph graph state
schema; its keys mirror :class:`TaskRecord` fields.
* :func:`new_thread_id` — uuid thread-id minting.
* JSON serialization helpers (:func:`task_to_dict` / :func:`task_from_dict` /
:func:`task_to_json` / :func:`task_from_json`).
The signatures here are CONTRACTS leaf builders import verbatim.
"""
from __future__ import annotations
import json
import uuid
from dataclasses import asdict, dataclass, field
from enum import Enum
from typing import Any, TypedDict
__all__ = [
"Phase",
"PipelineState",
"TaskRecord",
"TaskStatus",
"new_thread_id",
"task_from_dict",
"task_from_json",
"task_to_dict",
"task_to_json",
]
class TaskStatus(Enum):
"""Top-level task lifecycle status.
``ACTIVE`` — progressing through stages. ``WAITING_HUMAN`` — suspended on a
LangGraph ``interrupt()`` awaiting Adam's answer. ``PARKED`` — stalled
(no answer in window, N failed build loops, or budget contention) and
ALARM-ed rather than spinning (§3.3, §6.6). ``DONE`` — draft PR + report
produced. ``FAILED`` — terminal failure.
"""
ACTIVE = "active"
WAITING_HUMAN = "waiting_human"
PARKED = "parked"
DONE = "done"
FAILED = "failed"
class Phase(Enum):
"""Pipeline phase the task is currently in (§3.3)."""
INTAKE = "intake"
CLARIFY = "clarify"
PLAN = "plan"
REVIEW = "review"
BUILD = "build"
VERIFY = "verify"
PARKED = "parked"
DONE = "done"
def new_thread_id() -> str:
"""Mint a fresh unique ``thread_id`` (uuid4 hex)."""
return uuid.uuid4().hex
@dataclass
class TaskRecord:
"""The durable per-task record (§3.3, §3.3.1).
Mirrors the LangGraph thread state; the SQLite checkpointer persists the
graph state while this record is the logical view the coordinator reasons
over. ``qa_history`` is the full clarifier Q&A; ``review_verdicts`` the
adversarial review outcomes; ``candidate_diff`` + ``diff_hash`` the builder
output and its ledger-recorded hash (§3.3.2); ``ci_results`` the
authenticated CI conclusion the verifier reads.
"""
thread_id: str
status: TaskStatus
current_phase: Phase
# Intake task description (mirrors PipelineState.task).
task: str = ""
feat(agent-team): one Slack thread per task — root "Task received" message + threaded questions/milestones WS Slack-UX Feature 1. A /new-task task now maps to ONE Slack thread instead of several top-level messages. - /new-task posts an immediate root "📥 Task received: …" ack and captures its ts (root_ts); this is the instant acknowledgement. - root_ts is plumbed into start: new PipelineState/TaskRecord channel slack_thread_ts, seeded by graph.start_task and threaded through Coordinator.start_task. The NewTaskCallback is now (task_text, via, root_ts). - All clarifier questions for the task post as THREADED REPLIES under root_ts (chat.postMessage thread_ts=root_ts), and each question's ledger channel_ref is set to root_ts (NOT the reply's own ts). Because answer-mapping resolves a reply via find_open_question_by_channel_ref(thread_ts), a reply in the root thread (thread_ts==root_ts) maps to the task's currently-open question with NO change to the mapping logic or the first-answer-wins CAS. The open-only partial-unique index still holds (one open question per task at a time). - Lifecycle milestones (parked / plan-ready / needs-input) and follow-up questions thread under root_ts too; the notify sink gained an optional thread_ts kwarg (degrades to top-level on a sink that doesn't accept it). notify failures still never break tick. - SlackTransport.post_question + the live poster accept/forward thread_ts. - No root_ts (non-/new-task origin) ⇒ top-level posts exactly as before. AUTHZ-01 (owner-allowlist-first, fail-closed) and the atomic open→answered compare-and-set are unchanged. Adds plumbing for the inbound-ack reactor seam used by Feature 2 (dormant until a reactor is injected). Tests cover thread_ts forwarding, channel_ref=root_ts, graph seeding, and coordinator threading.
2026-06-23 15:11:19 -04:00
# Slack root-message ts for one-thread-per-task (mirrors
# PipelineState.slack_thread_ts). Empty for non-/new-task origins.
slack_thread_ts: str = ""
qa_history: list[Any] = field(default_factory=list)
plan: dict[str, Any] | None = None
review_verdicts: list[Any] = field(default_factory=list)
candidate_diff: str | None = None
diff_hash: str | None = None
ci_results: dict[str, Any] | None = None
# P3 box-side build->dispatch->verify plumbing (§3.3.2). Set by the DISPATCH
# node when it triggers the apply/verify CI run: ``run_id`` is the GitHub
# Actions run id the verifier's read-only fetcher polls + the pure-code gate
# binds its verdict to; ``ci_correlation_tag`` is the per-dispatch nonce
# carried as a workflow input so the run_id poll matches THIS task's exact
# run (anti-race / anti-replay); ``dispatched_at`` bounds the CI-watch
# timeout. All None until a task reaches DISPATCH on the live P3 path.
run_id: str | None = None
ci_correlation_tag: str | None = None
dispatched_at: str | None = None
fix(agent-team): remediate C1 security-review BLOCK (2 HIGH + MED/LOW) High-recall /sh-security-review fan-out + proof-or-kill verifier found two confirmed HIGH; both now closed (verified empirically against the working tree): - LOGIC-RACE-01 (HIGH, CWE-835): the build-loop budget was structurally dead (verifier read a shared wiring-time VerifierConfig.build_loops, always 0, so the max_build_loops park never fired -> a perpetually-failing task looped BUILD->DISPATCH->VERIFY forever, force-pushing + firing a CI run each round). Threaded build_loops through durable PipelineState/TaskRecord; verifier reads state.get('build_loops',0), writes the incremented count back on each FAIL, and PARKS at max_build_loops. Parks after exactly N failures, never unbounded. - SEC-01 (HIGH, CWE-532) + SEC-02 (MED, CWE-214): p3_rollback.sh echoed the live App JWT to stdout in default dry-run and passed it as a gh argv literal. Added redact_secrets (Bearer/Authorization/ghX_/PEM masking) through run_or_plan; the App uninstall now uses curl -H @<0600 tempfile> (JWT never on argv), shredded after. Empirical: app/incident/all dry-runs leak 0 JWT occurrences. - SEC-03 (MED, CWE-798): assert_no_write_token now applies the PEM regex + the configured App-ID to env/config VALUES (not just files) — an App private key under a benign env name is caught. - SEC-04 (LOW) + P3-IAC-08 (LOW): tightened the box GITHUB_TOKEN fallback / value-scan; staged-only WARN on the live workflow revert. Suite: 1382 passed, ruff clean. Branch only; not merged/deployed. NOTE: re-verifier flagged SEC-01 as open by grepping COMMITTED blobs (the fix was uncommitted working-tree state); independently confirmed closed empirically.
2026-06-23 19:40:59 -04:00
# Count of build<->verify loops already consumed for THIS task (§3.3 #6).
# Durable per-task state (NOT the shared wiring-time VerifierConfig): the
# verifier reads it from state, increments on a recoverable gate FAIL, and
# parks once it reaches VerifierConfig.max_build_loops so a perpetually-
# failing task can never loop BUILD->DISPATCH->VERIFY forever (LOGIC-RACE-01).
build_loops: int = 0
feat(agent-team): resumable plan-review gate in the graph (Phase B2a) Replace the terminal review-cap PARK with a resumable human decision gate, opt-in via build_graph(plan_gate=True) (default False → all existing P1/P2/P3 wiring unchanged). - plan_gate_node interrupt()s mirroring the clarifier contract (same payload keys → existing pending_question() extractor + turn-guarded ResumeWorker drive it with zero special-casing) plus a kind="plan_decision" discriminator and the plan + latest review findings as context. - Decision contract {"decision": approve|request_changes|abandon, "notes": ...}: approve → the same terminal state an auto-approved plan reaches (ACTIVE/BUILD); request_changes → append a synthetic human verdict to review_verdicts (so the planner's _format_review_feedback surfaces the notes) and loop back to PLAN; abandon / unrecognized → terminal FAILED (safe default, never accidental approve). - Bounded termination: MAX_PLAN_GATE_VISITS=3 combined ceiling on plan_gate_visits (new channel on PipelineState + TaskRecord); on exhaustion the gate goes terminal PARKED ("revision ceiling reached") WITHOUT interrupting. Proven by a loop-past-ceiling test. Notes for the coordinator wiring (B2b): while suspended at the gate the status channel still reads 'parked' (carried over from review_node's escalate branch) — the load-bearing "awaiting decision, not terminal" signal is the live pending interrupt + kind="plan_decision", NOT the status channel. Tests: interrupt-at-cap, approve/request_changes(notes)/abandon routing, ceiling-terminates, auto-approve still bypasses the gate. 1404 passed.
2026-06-23 20:11:32 -04:00
# Count of plan-review human-decision gate visits already consumed for THIS
# task (Phase B2a). The graph routes a review-cap dead-end to the plan gate,
# which interrupts for an owner approve / request_changes / abandon decision.
# A human ``request_changes`` re-enters plan<->review, which can hit the cap
# and gate AGAIN; this counter is the combined ceiling on human-driven gate
# loops (mirrors PipelineState.plan_gate_visits) so the loop always
# terminates: once it reaches ``MAX_PLAN_GATE_VISITS`` the gate stops
# offering request_changes and the task goes terminal PARKED.
plan_gate_visits: int = 0
transport: str = ""
created_at: str | None = None
updated_at: str | None = None
# Set when a pipeline node raised during resume and the coordinator failed
# the task (mirrors PipelineState.failure_reason). Empty on a healthy task.
failure_reason: str = ""
class PipelineState(TypedDict, total=False):
"""LangGraph graph-state schema; keys mirror :class:`TaskRecord` (§3.3).
Used as the graph's state type. ``total=False`` so a node may write a
subset of keys per checkpoint transition.
"""
thread_id: str
status: str
current_phase: str
# The intake task description (Slack /new-task text, GitHub issue body, etc.).
# Seeded by graph.start_task and read by the clarifier/planner; a first-class
# channel so the seeded value persists across node transitions.
task: str
feat(agent-team): one Slack thread per task — root "Task received" message + threaded questions/milestones WS Slack-UX Feature 1. A /new-task task now maps to ONE Slack thread instead of several top-level messages. - /new-task posts an immediate root "📥 Task received: …" ack and captures its ts (root_ts); this is the instant acknowledgement. - root_ts is plumbed into start: new PipelineState/TaskRecord channel slack_thread_ts, seeded by graph.start_task and threaded through Coordinator.start_task. The NewTaskCallback is now (task_text, via, root_ts). - All clarifier questions for the task post as THREADED REPLIES under root_ts (chat.postMessage thread_ts=root_ts), and each question's ledger channel_ref is set to root_ts (NOT the reply's own ts). Because answer-mapping resolves a reply via find_open_question_by_channel_ref(thread_ts), a reply in the root thread (thread_ts==root_ts) maps to the task's currently-open question with NO change to the mapping logic or the first-answer-wins CAS. The open-only partial-unique index still holds (one open question per task at a time). - Lifecycle milestones (parked / plan-ready / needs-input) and follow-up questions thread under root_ts too; the notify sink gained an optional thread_ts kwarg (degrades to top-level on a sink that doesn't accept it). notify failures still never break tick. - SlackTransport.post_question + the live poster accept/forward thread_ts. - No root_ts (non-/new-task origin) ⇒ top-level posts exactly as before. AUTHZ-01 (owner-allowlist-first, fail-closed) and the atomic open→answered compare-and-set are unchanged. Adds plumbing for the inbound-ack reactor seam used by Feature 2 (dormant until a reactor is injected). Tests cover thread_ts forwarding, channel_ref=root_ts, graph seeding, and coordinator threading.
2026-06-23 15:11:19 -04:00
# The Slack root-message ``ts`` for a /new-task task (the "📥 Task received"
# ack post). All of the task's clarifier questions and lifecycle milestone
# notifications thread under this ``ts`` so one task maps to one Slack thread.
# Empty/absent for a task that did not originate from /new-task (no root post),
# in which case posts are top-level exactly as before.
slack_thread_ts: str
qa_history: list[Any]
plan: dict[str, Any] | None
review_verdicts: list[Any]
candidate_diff: str | None
diff_hash: str | None
ci_results: dict[str, Any] | None
# P3 box-side build->dispatch->verify plumbing (mirrors TaskRecord). ``run_id``
# is the dispatched apply/verify Actions run id the verifier fetches + the
# gate binds to; ``ci_correlation_tag`` is the per-dispatch nonce carried as a
# workflow input so the poll matches THIS task's run; ``dispatched_at`` bounds
# the CI-watch timeout.
run_id: str | None
ci_correlation_tag: str | None
dispatched_at: str | None
fix(agent-team): remediate C1 security-review BLOCK (2 HIGH + MED/LOW) High-recall /sh-security-review fan-out + proof-or-kill verifier found two confirmed HIGH; both now closed (verified empirically against the working tree): - LOGIC-RACE-01 (HIGH, CWE-835): the build-loop budget was structurally dead (verifier read a shared wiring-time VerifierConfig.build_loops, always 0, so the max_build_loops park never fired -> a perpetually-failing task looped BUILD->DISPATCH->VERIFY forever, force-pushing + firing a CI run each round). Threaded build_loops through durable PipelineState/TaskRecord; verifier reads state.get('build_loops',0), writes the incremented count back on each FAIL, and PARKS at max_build_loops. Parks after exactly N failures, never unbounded. - SEC-01 (HIGH, CWE-532) + SEC-02 (MED, CWE-214): p3_rollback.sh echoed the live App JWT to stdout in default dry-run and passed it as a gh argv literal. Added redact_secrets (Bearer/Authorization/ghX_/PEM masking) through run_or_plan; the App uninstall now uses curl -H @<0600 tempfile> (JWT never on argv), shredded after. Empirical: app/incident/all dry-runs leak 0 JWT occurrences. - SEC-03 (MED, CWE-798): assert_no_write_token now applies the PEM regex + the configured App-ID to env/config VALUES (not just files) — an App private key under a benign env name is caught. - SEC-04 (LOW) + P3-IAC-08 (LOW): tightened the box GITHUB_TOKEN fallback / value-scan; staged-only WARN on the live workflow revert. Suite: 1382 passed, ruff clean. Branch only; not merged/deployed. NOTE: re-verifier flagged SEC-01 as open by grepping COMMITTED blobs (the fix was uncommitted working-tree state); independently confirmed closed empirically.
2026-06-23 19:40:59 -04:00
# Build<->verify loops already consumed for THIS task (mirrors TaskRecord).
# The verifier threads it through DURABLE state — increments on a recoverable
# gate FAIL, parks at VerifierConfig.max_build_loops — so the build-loop
# budget is real (LOGIC-RACE-01: it was previously read from the shared
# wiring-time config and never advanced).
build_loops: int
feat(agent-team): resumable plan-review gate in the graph (Phase B2a) Replace the terminal review-cap PARK with a resumable human decision gate, opt-in via build_graph(plan_gate=True) (default False → all existing P1/P2/P3 wiring unchanged). - plan_gate_node interrupt()s mirroring the clarifier contract (same payload keys → existing pending_question() extractor + turn-guarded ResumeWorker drive it with zero special-casing) plus a kind="plan_decision" discriminator and the plan + latest review findings as context. - Decision contract {"decision": approve|request_changes|abandon, "notes": ...}: approve → the same terminal state an auto-approved plan reaches (ACTIVE/BUILD); request_changes → append a synthetic human verdict to review_verdicts (so the planner's _format_review_feedback surfaces the notes) and loop back to PLAN; abandon / unrecognized → terminal FAILED (safe default, never accidental approve). - Bounded termination: MAX_PLAN_GATE_VISITS=3 combined ceiling on plan_gate_visits (new channel on PipelineState + TaskRecord); on exhaustion the gate goes terminal PARKED ("revision ceiling reached") WITHOUT interrupting. Proven by a loop-past-ceiling test. Notes for the coordinator wiring (B2b): while suspended at the gate the status channel still reads 'parked' (carried over from review_node's escalate branch) — the load-bearing "awaiting decision, not terminal" signal is the live pending interrupt + kind="plan_decision", NOT the status channel. Tests: interrupt-at-cap, approve/request_changes(notes)/abandon routing, ceiling-terminates, auto-approve still bypasses the gate. 1404 passed.
2026-06-23 20:11:32 -04:00
# Plan-review human-decision gate visits consumed for THIS task (mirrors
# TaskRecord.plan_gate_visits; Phase B2a). The combined ceiling on
# human-driven plan<->review loops: once it reaches MAX_PLAN_GATE_VISITS the
# gate stops offering request_changes and the task goes terminal PARKED, so
# the human-in-the-loop revision cycle can never spin forever.
plan_gate_visits: int
transport: str
created_at: str | None
updated_at: str | None
# Set when the coordinator terminally fails a task because a pipeline node
# raised during resume (see ``Coordinator._fail_resumed_task``). Carries a
# short "ExcType: message" so the failure notification can say what broke.
# Absent on a healthy task.
failure_reason: str
def task_to_dict(record: TaskRecord) -> dict[str, Any]:
"""Serialize a :class:`TaskRecord` to a JSON-safe dict (enums -> values)."""
data = asdict(record)
data["status"] = record.status.value
data["current_phase"] = record.current_phase.value
return data
def task_from_dict(data: dict[str, Any]) -> TaskRecord:
"""Rebuild a :class:`TaskRecord` from a :func:`task_to_dict` dict."""
return TaskRecord(
thread_id=data["thread_id"],
status=TaskStatus(data["status"]),
current_phase=Phase(data["current_phase"]),
task=data.get("task", ""),
feat(agent-team): one Slack thread per task — root "Task received" message + threaded questions/milestones WS Slack-UX Feature 1. A /new-task task now maps to ONE Slack thread instead of several top-level messages. - /new-task posts an immediate root "📥 Task received: …" ack and captures its ts (root_ts); this is the instant acknowledgement. - root_ts is plumbed into start: new PipelineState/TaskRecord channel slack_thread_ts, seeded by graph.start_task and threaded through Coordinator.start_task. The NewTaskCallback is now (task_text, via, root_ts). - All clarifier questions for the task post as THREADED REPLIES under root_ts (chat.postMessage thread_ts=root_ts), and each question's ledger channel_ref is set to root_ts (NOT the reply's own ts). Because answer-mapping resolves a reply via find_open_question_by_channel_ref(thread_ts), a reply in the root thread (thread_ts==root_ts) maps to the task's currently-open question with NO change to the mapping logic or the first-answer-wins CAS. The open-only partial-unique index still holds (one open question per task at a time). - Lifecycle milestones (parked / plan-ready / needs-input) and follow-up questions thread under root_ts too; the notify sink gained an optional thread_ts kwarg (degrades to top-level on a sink that doesn't accept it). notify failures still never break tick. - SlackTransport.post_question + the live poster accept/forward thread_ts. - No root_ts (non-/new-task origin) ⇒ top-level posts exactly as before. AUTHZ-01 (owner-allowlist-first, fail-closed) and the atomic open→answered compare-and-set are unchanged. Adds plumbing for the inbound-ack reactor seam used by Feature 2 (dormant until a reactor is injected). Tests cover thread_ts forwarding, channel_ref=root_ts, graph seeding, and coordinator threading.
2026-06-23 15:11:19 -04:00
slack_thread_ts=data.get("slack_thread_ts", ""),
qa_history=list(data.get("qa_history", [])),
plan=data.get("plan"),
review_verdicts=list(data.get("review_verdicts", [])),
candidate_diff=data.get("candidate_diff"),
diff_hash=data.get("diff_hash"),
ci_results=data.get("ci_results"),
run_id=data.get("run_id"),
ci_correlation_tag=data.get("ci_correlation_tag"),
dispatched_at=data.get("dispatched_at"),
fix(agent-team): remediate C1 security-review BLOCK (2 HIGH + MED/LOW) High-recall /sh-security-review fan-out + proof-or-kill verifier found two confirmed HIGH; both now closed (verified empirically against the working tree): - LOGIC-RACE-01 (HIGH, CWE-835): the build-loop budget was structurally dead (verifier read a shared wiring-time VerifierConfig.build_loops, always 0, so the max_build_loops park never fired -> a perpetually-failing task looped BUILD->DISPATCH->VERIFY forever, force-pushing + firing a CI run each round). Threaded build_loops through durable PipelineState/TaskRecord; verifier reads state.get('build_loops',0), writes the incremented count back on each FAIL, and PARKS at max_build_loops. Parks after exactly N failures, never unbounded. - SEC-01 (HIGH, CWE-532) + SEC-02 (MED, CWE-214): p3_rollback.sh echoed the live App JWT to stdout in default dry-run and passed it as a gh argv literal. Added redact_secrets (Bearer/Authorization/ghX_/PEM masking) through run_or_plan; the App uninstall now uses curl -H @<0600 tempfile> (JWT never on argv), shredded after. Empirical: app/incident/all dry-runs leak 0 JWT occurrences. - SEC-03 (MED, CWE-798): assert_no_write_token now applies the PEM regex + the configured App-ID to env/config VALUES (not just files) — an App private key under a benign env name is caught. - SEC-04 (LOW) + P3-IAC-08 (LOW): tightened the box GITHUB_TOKEN fallback / value-scan; staged-only WARN on the live workflow revert. Suite: 1382 passed, ruff clean. Branch only; not merged/deployed. NOTE: re-verifier flagged SEC-01 as open by grepping COMMITTED blobs (the fix was uncommitted working-tree state); independently confirmed closed empirically.
2026-06-23 19:40:59 -04:00
build_loops=data.get("build_loops", 0),
feat(agent-team): resumable plan-review gate in the graph (Phase B2a) Replace the terminal review-cap PARK with a resumable human decision gate, opt-in via build_graph(plan_gate=True) (default False → all existing P1/P2/P3 wiring unchanged). - plan_gate_node interrupt()s mirroring the clarifier contract (same payload keys → existing pending_question() extractor + turn-guarded ResumeWorker drive it with zero special-casing) plus a kind="plan_decision" discriminator and the plan + latest review findings as context. - Decision contract {"decision": approve|request_changes|abandon, "notes": ...}: approve → the same terminal state an auto-approved plan reaches (ACTIVE/BUILD); request_changes → append a synthetic human verdict to review_verdicts (so the planner's _format_review_feedback surfaces the notes) and loop back to PLAN; abandon / unrecognized → terminal FAILED (safe default, never accidental approve). - Bounded termination: MAX_PLAN_GATE_VISITS=3 combined ceiling on plan_gate_visits (new channel on PipelineState + TaskRecord); on exhaustion the gate goes terminal PARKED ("revision ceiling reached") WITHOUT interrupting. Proven by a loop-past-ceiling test. Notes for the coordinator wiring (B2b): while suspended at the gate the status channel still reads 'parked' (carried over from review_node's escalate branch) — the load-bearing "awaiting decision, not terminal" signal is the live pending interrupt + kind="plan_decision", NOT the status channel. Tests: interrupt-at-cap, approve/request_changes(notes)/abandon routing, ceiling-terminates, auto-approve still bypasses the gate. 1404 passed.
2026-06-23 20:11:32 -04:00
plan_gate_visits=data.get("plan_gate_visits", 0),
transport=data.get("transport", ""),
created_at=data.get("created_at"),
updated_at=data.get("updated_at"),
)
def task_to_json(record: TaskRecord) -> str:
"""Serialize a :class:`TaskRecord` to a JSON string."""
return json.dumps(task_to_dict(record), sort_keys=True)
def task_from_json(payload: str | bytes) -> TaskRecord:
"""Deserialize a :class:`TaskRecord` from a JSON string/bytes."""
return task_from_dict(json.loads(payload))