This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/agent_team/db/transitions.py
Adam Moussa 6951bf2fc6 fix(agent-team): address GPT-4.1 cross-review findings on the dashboard surface
- read_transitions: catch sqlite3.Error (not just OperationalError) so a corrupt
  ledger degrades to empty rather than raising into callers
- recorder open-row lookup: order by the monotonic transition_id (drop the
  timestamp-format dependency)
- TransitionRecorder.close(): release the retained in-memory test connection
- dashboard task_detail: return only the exception TYPE, never str(exc) (a SQLite
  message can carry the DB path)
- _instrument: coerce a status enum to .value defensively before the terminal check
- schema.migrate: document the ordering constraint for future ALTERs vs the
  unconditional idempotent tail
2026-06-23 17:19:04 -04:00

206 lines
8.3 KiB
Python

"""Per-task pipeline transition recorder (schema v3, ``task_transitions``).
The coordinator wraps each LangGraph node (``graph._instrument``) so that, as a
task enters a node, one row is appended to ``task_transitions`` capturing the
move ``from_phase -> to_phase`` with an ``entered_at`` stamp and the task status
at entry. The dashboard's ``/api/task/{thread_id}`` drill-down reads these rows
to render a task's journey through the pipeline; per-stage cost is joined from
``budget_ledger`` at read time and is NOT duplicated here.
Two design properties matter (both exercised by tests):
* **Fail-soft.** Every write is wrapped so a ledger problem (locked DB, missing
table, disk error) is logged and swallowed — instrumentation must NEVER break
the live pipeline. A dropped transition row degrades the dashboard, nothing
more.
* **Idempotent under replay.** LangGraph re-executes a node from its start on
resume (e.g. the clarifier replays after the human gate). So a node's wrapper
may call :meth:`TransitionRecorder.record_entry` more than once for the same
``(thread_id, to_phase)``. The recorder keeps a single OPEN row per thread:
re-entering the *same* node while it is already the open row is a no-op;
entering a *different* node closes the previous open row (filling its
``exited_at``) before inserting the new one. A crash/resume therefore cannot
leave orphaned open rows or double-count a replayed node.
"""
from __future__ import annotations
import logging
import sqlite3
from datetime import datetime, timezone
from pathlib import Path
from agent_team.db.schema import connect
__all__ = ["TransitionRecorder", "read_transitions"]
_log = logging.getLogger(__name__)
def _utc_now_iso() -> str:
"""Current UTC time as an ISO-8601 string (matches the other ledgers)."""
return datetime.now(timezone.utc).isoformat()
def read_transitions(
conn: sqlite3.Connection, thread_id: str
) -> list[dict[str, object]]:
"""Return ``thread_id``'s transition rows in entry order (oldest first).
Read-only and parameterized; callers pass their own connection (the
dashboard uses a strictly read-only one). Returns ``[]`` if the table does
not exist yet (fresh DB) rather than raising, so a status view never crashes.
"""
try:
rows = conn.execute(
"SELECT transition_id, thread_id, from_phase, to_phase, "
"entered_at, exited_at, status, note "
"FROM task_transitions WHERE thread_id = ? "
"ORDER BY entered_at ASC, transition_id ASC",
(thread_id,),
).fetchall()
except sqlite3.Error:
# Missing table (fresh DB) or a corrupt/unreadable ledger — a status
# view must never crash on a read.
return []
return [dict(row) for row in rows]
class TransitionRecorder:
"""Fail-soft writer for ``task_transitions`` (one OPEN row per thread).
Constructed with the ledger DB **path** (not a connection): the coordinator
owns the writable side, and each call opens a short-lived WAL connection via
:func:`agent_team.db.schema.connect`, mirroring the compare-and-set helpers'
connection discipline. An in-memory path (``":memory:"``) is supported for
tests by reusing a single retained connection (a fresh ``:memory:`` connect
would see an empty database).
"""
def __init__(self, db_path: Path | str) -> None:
self._db_path = Path(db_path)
self._in_memory = str(db_path) == ":memory:"
# In-memory DBs are per-connection; retain one so writes accumulate.
self._mem_conn: sqlite3.Connection | None = (
connect(self._db_path) if self._in_memory else None
)
def _connect(self) -> sqlite3.Connection:
if self._mem_conn is not None:
return self._mem_conn
return connect(self._db_path)
def _close(self, conn: sqlite3.Connection) -> None:
# Never close the retained in-memory connection.
if conn is not self._mem_conn:
conn.close()
def close(self) -> None:
"""Close the retained in-memory connection, if any (test cleanup).
File-backed recorders open/close per call and hold nothing, so this is a
no-op for them; the in-memory test path retains one connection that this
releases so it does not leak when the recorder is discarded.
"""
if self._mem_conn is not None:
self._mem_conn.close()
self._mem_conn = None
def record_entry(
self,
*,
thread_id: str,
to_phase: str,
status: str | None = None,
note: str | None = None,
) -> None:
"""Append an OPEN transition for entering ``to_phase`` (idempotent).
If the thread's latest open row is already this ``to_phase`` (a resume
replay of the same node), this is a no-op. If the latest open row is a
*different* node, it is closed (``exited_at`` filled) before the new open
row is inserted, carrying that node's ``to_phase`` forward as the new
row's ``from_phase``. Fail-soft: any error is logged and swallowed.
"""
if not thread_id or not to_phase:
return
conn = None
try:
conn = self._connect()
now = _utc_now_iso()
open_row = conn.execute(
"SELECT transition_id, to_phase FROM task_transitions "
"WHERE thread_id = ? AND exited_at IS NULL "
"ORDER BY transition_id DESC LIMIT 1",
(thread_id,),
).fetchone()
from_phase: str | None = None
if open_row is not None:
if open_row["to_phase"] == to_phase:
# Same node re-entered (resume replay) — no double row.
return
# Different node: close the previous open row.
conn.execute(
"UPDATE task_transitions SET exited_at = ? WHERE transition_id = ?",
(now, open_row["transition_id"]),
)
from_phase = open_row["to_phase"]
conn.execute(
"INSERT INTO task_transitions "
"(thread_id, from_phase, to_phase, entered_at, exited_at, "
"status, note) VALUES (?, ?, ?, ?, NULL, ?, ?)",
(thread_id, from_phase, to_phase, now, status, note),
)
except Exception as exc: # fail-soft: never break the pipeline
_log.warning("task_transitions record_entry failed: %s", exc)
finally:
if conn is not None:
self._close(conn)
def close_terminal(
self,
*,
thread_id: str,
status: str | None = None,
note: str | None = None,
) -> None:
"""Close the thread's open transition on a terminal status (N2).
The last node a task visits never gets an ``exited_at`` from a *next*
transition, so when a node returns a terminal status (DONE/PARKED/FAILED)
the wrapper calls this to stamp ``exited_at`` (and optionally update the
row's ``status``/``note``). No-op if there is no open row. Fail-soft.
"""
if not thread_id:
return
conn = None
try:
conn = self._connect()
now = _utc_now_iso()
open_row = conn.execute(
"SELECT transition_id FROM task_transitions "
"WHERE thread_id = ? AND exited_at IS NULL "
"ORDER BY transition_id DESC LIMIT 1",
(thread_id,),
).fetchone()
if open_row is None:
return
if status is not None:
conn.execute(
"UPDATE task_transitions SET exited_at = ?, status = ?, "
"note = COALESCE(?, note) WHERE transition_id = ?",
(now, status, note, open_row["transition_id"]),
)
else:
conn.execute(
"UPDATE task_transitions SET exited_at = ?, "
"note = COALESCE(?, note) WHERE transition_id = ?",
(now, note, open_row["transition_id"]),
)
except Exception as exc: # fail-soft
_log.warning("task_transitions close_terminal failed: %s", exc)
finally:
if conn is not None:
self._close(conn)