This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/agent_team/task_model.py
Adam Moussa f12dfecd95 fix(agent-team): a crashing pipeline node fails one task, not the whole daemon
A live task (d30b697c) on the R720 crashed the coordinator: the planner's
single-shot Claude call raised "Reached maximum number of turns (1)", the
exception propagated out of `drain_resumes` through the serve loop, and systemd
restarted the daemon — with no Slack notice, so the failure was silent.

Defense in depth:

1. invoker: `_collect_subscription_text` now tolerates the single-shot turn cap.
   When the Agent SDK raises "Reached maximum number of turns" mid-stream it
   salvages the assistant text already collected (the JSON the planner needs)
   instead of propagating. An empty salvage or any non-turn error still raises.

2. coordinator: `drain_resumes` wraps the per-job `resume()` so an unhandled
   node exception fails THAT task instead of the daemon — it supersedes the
   answered question (so the startup recovery sweep cannot re-drive it into the
   same crash on reboot), marks the task FAILED via `update_state` with a short
   `failure_reason`, and surfaces an honest "❌ FAILED" line to Slack.

3. task_model: add `failure_reason` to PipelineState + TaskRecord (kept in sync)
   so the terminal-failure detail persists as a real graph channel.

4. resume_worker: add `ResumeOutcome.FAILED`.

Tests: invoker salvage/re-raise/propagate paths; drain_resumes fails-not-crashes,
supersedes the question, notifies, and one failing task does not block others.
1189 passed.
2026-06-23 16:20:45 -04:00

180 lines
6.3 KiB
Python

"""Task-record / thread model + LangGraph graph-state schema (design §3.3).
A task is a long-lived, resumable record (a LangGraph thread). This module is
the pure model layer — no I/O — defining:
* :class:`TaskStatus` / :class:`Phase` — task lifecycle enums.
* :class:`TaskRecord` — the durable task record (§3.3 "the task record holds:
status, current phase, the full Q&A history, the plan, review verdicts, the
candidate diff, and CI results").
* :class:`PipelineState` — a ``TypedDict`` used as the LangGraph graph state
schema; its keys mirror :class:`TaskRecord` fields.
* :func:`new_thread_id` — uuid thread-id minting.
* JSON serialization helpers (:func:`task_to_dict` / :func:`task_from_dict` /
:func:`task_to_json` / :func:`task_from_json`).
The signatures here are CONTRACTS leaf builders import verbatim.
"""
from __future__ import annotations
import json
import uuid
from dataclasses import asdict, dataclass, field
from enum import Enum
from typing import Any, TypedDict
__all__ = [
"Phase",
"PipelineState",
"TaskRecord",
"TaskStatus",
"new_thread_id",
"task_from_dict",
"task_from_json",
"task_to_dict",
"task_to_json",
]
class TaskStatus(Enum):
"""Top-level task lifecycle status.
``ACTIVE`` — progressing through stages. ``WAITING_HUMAN`` — suspended on a
LangGraph ``interrupt()`` awaiting Adam's answer. ``PARKED`` — stalled
(no answer in window, N failed build loops, or budget contention) and
ALARM-ed rather than spinning (§3.3, §6.6). ``DONE`` — draft PR + report
produced. ``FAILED`` — terminal failure.
"""
ACTIVE = "active"
WAITING_HUMAN = "waiting_human"
PARKED = "parked"
DONE = "done"
FAILED = "failed"
class Phase(Enum):
"""Pipeline phase the task is currently in (§3.3)."""
INTAKE = "intake"
CLARIFY = "clarify"
PLAN = "plan"
REVIEW = "review"
BUILD = "build"
VERIFY = "verify"
PARKED = "parked"
DONE = "done"
def new_thread_id() -> str:
"""Mint a fresh unique ``thread_id`` (uuid4 hex)."""
return uuid.uuid4().hex
@dataclass
class TaskRecord:
"""The durable per-task record (§3.3, §3.3.1).
Mirrors the LangGraph thread state; the SQLite checkpointer persists the
graph state while this record is the logical view the coordinator reasons
over. ``qa_history`` is the full clarifier Q&A; ``review_verdicts`` the
adversarial review outcomes; ``candidate_diff`` + ``diff_hash`` the builder
output and its ledger-recorded hash (§3.3.2); ``ci_results`` the
authenticated CI conclusion the verifier reads.
"""
thread_id: str
status: TaskStatus
current_phase: Phase
# Intake task description (mirrors PipelineState.task).
task: str = ""
# Slack root-message ts for one-thread-per-task (mirrors
# PipelineState.slack_thread_ts). Empty for non-/new-task origins.
slack_thread_ts: str = ""
qa_history: list[Any] = field(default_factory=list)
plan: dict[str, Any] | None = None
review_verdicts: list[Any] = field(default_factory=list)
candidate_diff: str | None = None
diff_hash: str | None = None
ci_results: dict[str, Any] | None = None
transport: str = ""
created_at: str | None = None
updated_at: str | None = None
# Set when a pipeline node raised during resume and the coordinator failed
# the task (mirrors PipelineState.failure_reason). Empty on a healthy task.
failure_reason: str = ""
class PipelineState(TypedDict, total=False):
"""LangGraph graph-state schema; keys mirror :class:`TaskRecord` (§3.3).
Used as the graph's state type. ``total=False`` so a node may write a
subset of keys per checkpoint transition.
"""
thread_id: str
status: str
current_phase: str
# The intake task description (Slack /new-task text, GitHub issue body, etc.).
# Seeded by graph.start_task and read by the clarifier/planner; a first-class
# channel so the seeded value persists across node transitions.
task: str
# The Slack root-message ``ts`` for a /new-task task (the "📥 Task received"
# ack post). All of the task's clarifier questions and lifecycle milestone
# notifications thread under this ``ts`` so one task maps to one Slack thread.
# Empty/absent for a task that did not originate from /new-task (no root post),
# in which case posts are top-level exactly as before.
slack_thread_ts: str
qa_history: list[Any]
plan: dict[str, Any] | None
review_verdicts: list[Any]
candidate_diff: str | None
diff_hash: str | None
ci_results: dict[str, Any] | None
transport: str
created_at: str | None
updated_at: str | None
# Set when the coordinator terminally fails a task because a pipeline node
# raised during resume (see ``Coordinator._fail_resumed_task``). Carries a
# short "ExcType: message" so the failure notification can say what broke.
# Absent on a healthy task.
failure_reason: str
def task_to_dict(record: TaskRecord) -> dict[str, Any]:
"""Serialize a :class:`TaskRecord` to a JSON-safe dict (enums -> values)."""
data = asdict(record)
data["status"] = record.status.value
data["current_phase"] = record.current_phase.value
return data
def task_from_dict(data: dict[str, Any]) -> TaskRecord:
"""Rebuild a :class:`TaskRecord` from a :func:`task_to_dict` dict."""
return TaskRecord(
thread_id=data["thread_id"],
status=TaskStatus(data["status"]),
current_phase=Phase(data["current_phase"]),
task=data.get("task", ""),
slack_thread_ts=data.get("slack_thread_ts", ""),
qa_history=list(data.get("qa_history", [])),
plan=data.get("plan"),
review_verdicts=list(data.get("review_verdicts", [])),
candidate_diff=data.get("candidate_diff"),
diff_hash=data.get("diff_hash"),
ci_results=data.get("ci_results"),
transport=data.get("transport", ""),
created_at=data.get("created_at"),
updated_at=data.get("updated_at"),
)
def task_to_json(record: TaskRecord) -> str:
"""Serialize a :class:`TaskRecord` to a JSON string."""
return json.dumps(task_to_dict(record), sort_keys=True)
def task_from_json(payload: str | bytes) -> TaskRecord:
"""Deserialize a :class:`TaskRecord` from a JSON string/bytes."""
return task_from_dict(json.loads(payload))