"""Per-task pipeline transition recorder (schema v3, ``task_transitions``). The coordinator wraps each LangGraph node (``graph._instrument``) so that, as a task enters a node, one row is appended to ``task_transitions`` capturing the move ``from_phase -> to_phase`` with an ``entered_at`` stamp and the task status at entry. The dashboard's ``/api/task/{thread_id}`` drill-down reads these rows to render a task's journey through the pipeline; per-stage cost is joined from ``budget_ledger`` at read time and is NOT duplicated here. Two design properties matter (both exercised by tests): * **Fail-soft.** Every write is wrapped so a ledger problem (locked DB, missing table, disk error) is logged and swallowed — instrumentation must NEVER break the live pipeline. A dropped transition row degrades the dashboard, nothing more. * **Idempotent under replay.** LangGraph re-executes a node from its start on resume (e.g. the clarifier replays after the human gate). So a node's wrapper may call :meth:`TransitionRecorder.record_entry` more than once for the same ``(thread_id, to_phase)``. The recorder keeps a single OPEN row per thread: re-entering the *same* node while it is already the open row is a no-op; entering a *different* node closes the previous open row (filling its ``exited_at``) before inserting the new one. A crash/resume therefore cannot leave orphaned open rows or double-count a replayed node. """ from __future__ import annotations import logging import sqlite3 from datetime import datetime, timezone from pathlib import Path from agent_team.db.schema import connect __all__ = ["TransitionRecorder", "read_transitions"] _log = logging.getLogger(__name__) def _utc_now_iso() -> str: """Current UTC time as an ISO-8601 string (matches the other ledgers).""" return datetime.now(timezone.utc).isoformat() def read_transitions( conn: sqlite3.Connection, thread_id: str ) -> list[dict[str, object]]: """Return ``thread_id``'s transition rows in entry order (oldest first). Read-only and parameterized; callers pass their own connection (the dashboard uses a strictly read-only one). Returns ``[]`` if the table does not exist yet (fresh DB) rather than raising, so a status view never crashes. """ try: rows = conn.execute( "SELECT transition_id, thread_id, from_phase, to_phase, " "entered_at, exited_at, status, note " "FROM task_transitions WHERE thread_id = ? " "ORDER BY entered_at ASC, transition_id ASC", (thread_id,), ).fetchall() except sqlite3.Error: # Missing table (fresh DB) or a corrupt/unreadable ledger — a status # view must never crash on a read. return [] return [dict(row) for row in rows] class TransitionRecorder: """Fail-soft writer for ``task_transitions`` (one OPEN row per thread). Constructed with the ledger DB **path** (not a connection): the coordinator owns the writable side, and each call opens a short-lived WAL connection via :func:`agent_team.db.schema.connect`, mirroring the compare-and-set helpers' connection discipline. An in-memory path (``":memory:"``) is supported for tests by reusing a single retained connection (a fresh ``:memory:`` connect would see an empty database). """ def __init__(self, db_path: Path | str) -> None: self._db_path = Path(db_path) self._in_memory = str(db_path) == ":memory:" # In-memory DBs are per-connection; retain one so writes accumulate. self._mem_conn: sqlite3.Connection | None = ( connect(self._db_path) if self._in_memory else None ) def _connect(self) -> sqlite3.Connection: if self._mem_conn is not None: return self._mem_conn return connect(self._db_path) def _close(self, conn: sqlite3.Connection) -> None: # Never close the retained in-memory connection. if conn is not self._mem_conn: conn.close() def close(self) -> None: """Close the retained in-memory connection, if any (test cleanup). File-backed recorders open/close per call and hold nothing, so this is a no-op for them; the in-memory test path retains one connection that this releases so it does not leak when the recorder is discarded. """ if self._mem_conn is not None: self._mem_conn.close() self._mem_conn = None def record_entry( self, *, thread_id: str, to_phase: str, status: str | None = None, note: str | None = None, ) -> None: """Append an OPEN transition for entering ``to_phase`` (idempotent). If the thread's latest open row is already this ``to_phase`` (a resume replay of the same node), this is a no-op. If the latest open row is a *different* node, it is closed (``exited_at`` filled) before the new open row is inserted, carrying that node's ``to_phase`` forward as the new row's ``from_phase``. Fail-soft: any error is logged and swallowed. """ if not thread_id or not to_phase: return conn = None try: conn = self._connect() now = _utc_now_iso() open_row = conn.execute( "SELECT transition_id, to_phase FROM task_transitions " "WHERE thread_id = ? AND exited_at IS NULL " "ORDER BY transition_id DESC LIMIT 1", (thread_id,), ).fetchone() from_phase: str | None = None if open_row is not None: if open_row["to_phase"] == to_phase: # Same node re-entered (resume replay) — no double row. return # Different node: close the previous open row. conn.execute( "UPDATE task_transitions SET exited_at = ? WHERE transition_id = ?", (now, open_row["transition_id"]), ) from_phase = open_row["to_phase"] conn.execute( "INSERT INTO task_transitions " "(thread_id, from_phase, to_phase, entered_at, exited_at, " "status, note) VALUES (?, ?, ?, ?, NULL, ?, ?)", (thread_id, from_phase, to_phase, now, status, note), ) except Exception as exc: # fail-soft: never break the pipeline _log.warning("task_transitions record_entry failed: %s", exc) finally: if conn is not None: self._close(conn) def close_terminal( self, *, thread_id: str, status: str | None = None, note: str | None = None, ) -> None: """Close the thread's open transition on a terminal status (N2). The last node a task visits never gets an ``exited_at`` from a *next* transition, so when a node returns a terminal status (DONE/PARKED/FAILED) the wrapper calls this to stamp ``exited_at`` (and optionally update the row's ``status``/``note``). No-op if there is no open row. Fail-soft. """ if not thread_id: return conn = None try: conn = self._connect() now = _utc_now_iso() open_row = conn.execute( "SELECT transition_id FROM task_transitions " "WHERE thread_id = ? AND exited_at IS NULL " "ORDER BY transition_id DESC LIMIT 1", (thread_id,), ).fetchone() if open_row is None: return if status is not None: conn.execute( "UPDATE task_transitions SET exited_at = ?, status = ?, " "note = COALESCE(?, note) WHERE transition_id = ?", (now, status, note, open_row["transition_id"]), ) else: conn.execute( "UPDATE task_transitions SET exited_at = ?, " "note = COALESCE(?, note) WHERE transition_id = ?", (now, note, open_row["transition_id"]), ) except Exception as exc: # fail-soft _log.warning("task_transitions close_terminal failed: %s", exc) finally: if conn is not None: self._close(conn)