This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/agent_team/nodes/confluence_writer_llm.py
Adam Moussa c7f9c1bac2 feat(agent-team): Confluence-writer node (draft -> approve gate -> write)
Add a Confluence documentation lane to the Plane-2 pipeline, flag-gated behind
AGENT_TEAM_CONFLUENCE_ENABLED (default off; daemon behavior unchanged when off).

- confluence/client.py: OAuth 2LO + Basic REST client, dry-run-default writes
- confluence/mermaid.py: vendored ADF-only Mermaid editor (macro-count +
  revert-diff guards, dry-run default)
- nodes/confluence_writer.py(+_llm): conf_draft -> conf_gate -> conf_write,
  both direct (task_kind=confluence) and post-build documentation flows
- task_model/graph/coordinator: new phases, state channels, route_after_intake,
  CONFLUENCE_APPROVAL_KIND gate delivery, task_kind forwarding
- db schema v5: widen pending_questions kind CHECK (atomic rebuild)
- tests for client, mermaid, writer node, ledger v5, coordinator gate, e2e
2026-06-25 10:56:48 -04:00

384 lines
14 KiB
Python

"""Prompt-build / reply-parse split for the Confluence-writer node.
This module is the pure, model-output-facing half of the Confluence-writer
stage — the same ``build_*_prompt`` / ``parse_*_reply`` split
:mod:`agent_team.nodes.planner` (``build_plan_prompt`` / ``parse_plan``) and
:mod:`agent_team.nodes.clarifier_llm` (``ClaudeClarifier._build_prompt`` /
``_parse``) use, so the node body (:mod:`agent_team.nodes.confluence_writer`)
stays a thin LangGraph state-transition wrapper.
The Confluence-writer node drafts a **Confluence page update** as DATA (it never
writes to a repo, and it never posts to Confluence from the draft node — the
write is a later, dry-run-by-default node). Two source flows feed the draft:
* **Flow A** — a direct documentation ask: the draft is built from the task
description alone (``state['task']``).
* **Flow B** — a post-build documentation update: the draft folds in the
approved ``plan``, the ``candidate_diff``, and the repo name so the page
reflects what the change actually did.
On a ``request_changes`` loop-back the prior gate's ``confluence_feedback`` is
folded into the prompt so the redraft answers the reviewer's notes (mirrors the
planner's ``_format_review_feedback`` loop-back).
The LLM binding is an INJECTED seam (``context_provider``), exactly like the
planner's, so the prompt assembly stays pure and unit-testable and no live model
is bound here.
Parsing is DEFENSIVE: the model output is UNTRUSTED. A garbled reply must not
silently produce an empty page write, so :func:`parse_confluence_reply` raises
:class:`ConfluenceDraftError` when it cannot recover a usable
``{title, body_storage}`` draft, and the node decides what to do with that.
"""
from __future__ import annotations
import json
import re
from collections.abc import Callable, Mapping, Sequence
from typing import Any
from agent_team.task_model import PipelineState
# Optional context-provider callable (mirrors planner.ContextProvider): () -> str.
# When injected, its result is prepended to the Confluence-writer prompt. Default
# None = unchanged behavior so existing callers are unaffected.
ContextProvider = Callable[[], str]
__all__ = [
"ConfluenceDraftError",
"build_confluence_prompt",
"parse_confluence_reply",
]
class ConfluenceDraftError(Exception):
"""Raised when the model reply cannot be parsed into a usable draft.
Distinct from a *gate rejection* (which loops back through the human gate):
this signals the draft model output itself was empty/garbled, so the node
can fail loudly rather than advance an empty page update toward a write.
"""
def _task_description(state: PipelineState) -> str:
"""Pull the task description out of the graph state (mirrors planner.py)."""
plan = state.get("plan") or {}
if isinstance(plan, dict):
desc = plan.get("task") or plan.get("description")
if isinstance(desc, str) and desc.strip():
return desc.strip()
desc = state.get("task") # type: ignore[call-overload]
if isinstance(desc, str) and desc.strip():
return desc.strip()
return ""
def _repo_name(state: PipelineState) -> str:
"""Pull the repo name out of the graph state (Flow B), defensively.
Looked for under a top-level ``repo`` key and, failing that, inside the plan
dict — the same conventional places the clarifier reads ``repo`` from. An
absent repo yields an empty string so the prompt section is simply omitted.
"""
repo = state.get("repo") # type: ignore[call-overload]
if isinstance(repo, str) and repo.strip():
return repo.strip()
plan = state.get("plan") or {}
if isinstance(plan, dict):
repo = plan.get("repo")
if isinstance(repo, str) and repo.strip():
return repo.strip()
return ""
def _candidate_diff(state: PipelineState) -> str:
"""Pull the candidate diff (Flow B) out of the graph state, defensively."""
diff = state.get("candidate_diff")
if isinstance(diff, str) and diff.strip():
return diff.strip()
return ""
def _format_plan(plan: Any) -> str:
"""Render the approved plan into prompt text (Flow B)."""
if not isinstance(plan, Mapping):
return ""
try:
return json.dumps(dict(plan), sort_keys=True, indent=2)
except (TypeError, ValueError):
return repr(plan)
def _truncate(text: str, limit: int) -> str:
"""Truncate ``text`` to ``limit`` chars with a marker (keep prompts bounded).
A candidate diff can be large; the draft only needs enough of it to describe
the change, so we cap it rather than blow the context budget.
"""
if len(text) <= limit:
return text
return text[:limit] + f"\n... [truncated {len(text) - limit} chars]"
def build_confluence_prompt(
state: PipelineState,
*,
context_provider: ContextProvider | None = None,
) -> str:
"""Build the Claude prompt that drafts a Confluence page update.
Pure string assembly over the graph state — no I/O — so the prompt shape is
directly unit-testable (mirrors :func:`planner.build_plan_prompt`).
Source selection:
* **Flow A** (no plan/diff) — the draft is built from ``state['task']``.
* **Flow B** (plan and/or candidate diff present) — the draft folds in the
approved plan, the candidate diff, and the repo name.
On a ``request_changes`` loop-back the prior ``confluence_feedback`` is
folded in so the redraft addresses every note (mirrors the planner's
review-feedback loop-back). ``context_provider`` is the optional injection
seam: when set it is called once and its result prepended (and a raising
provider degrades to an empty prefix, never a crash).
"""
description = _task_description(state)
repo = _repo_name(state)
diff = _candidate_diff(state)
plan = state.get("plan")
plan_text = _format_plan(plan)
feedback = state.get("confluence_feedback")
flow_b = bool(plan_text or diff)
sections: list[str] = []
if context_provider is not None:
try:
ctx = context_provider()
except Exception: # noqa: BLE001 - a context provider must never crash the draft
ctx = ""
if ctx:
sections += [ctx, ""]
sections += [
"You are the CONFLUENCE-WRITER stage of an agentic SDLC pipeline. "
"Inspect the repository (read-only) and draft a Confluence page update "
"that documents this work. Emit the draft as DATA only — do NOT edit, "
"create, or write any files, and do NOT post to Confluence; a later "
"human-gated node performs the actual write.",
"",
"## Task",
description or "(no task description provided)",
]
if repo:
sections += ["", "## Repository", repo]
if flow_b:
if plan_text:
sections += ["", "## Approved plan (what was decided)", plan_text]
if diff:
sections += [
"",
"## Candidate diff (what the change actually did)",
_truncate(diff, 8000),
]
sections += [
"",
"## Documentation intent",
"Update the Confluence page so it reflects the architecture AFTER "
"this change: new/removed/modified resources, data flow, and "
"configuration. If the page carries an architecture diagram macro, "
"describe the mermaid edits needed rather than rewriting unrelated "
"content.",
]
else:
sections += [
"",
"## Documentation intent",
"Draft the Confluence page update this documentation task asks for.",
]
if isinstance(feedback, str) and feedback.strip():
sections += [
"",
"## Reviewer feedback on the previous draft (address every point)",
feedback.strip(),
]
sections += [
"",
"## Output format",
"Return ONLY a JSON object with keys: "
'"title" (string, the page title), '
'"body_storage" (string, the page body in Confluence STORAGE format / '
"XHTML), "
'"page_id" (optional string, the id of an existing page to update; omit '
"or null to create), and "
'"mermaid_edits" (optional list of objects each with "macro_id" or '
'"anchor" and a "mermaid" string, for architecture-diagram macro '
"updates). Do not include prose outside the JSON. Do NOT use any "
"write/edit tools.",
]
return "\n".join(sections)
# A fenced ```...``` block, if the model wrapped its JSON in Markdown.
_FENCE_RE = re.compile(
r"```(?:json)?\s*\n?(?P<body>.*?)\n?\s*```",
flags=re.DOTALL | re.IGNORECASE,
)
def _strip_code_fence(text: str) -> str:
"""Strip a leading/trailing Markdown code fence if present (mirrors planner)."""
fenced = re.match(
r"^\s*```(?:json)?\s*\n(?P<body>.*?)\n?\s*```\s*$",
text,
flags=re.DOTALL | re.IGNORECASE,
)
if fenced:
return fenced.group("body")
return text
def _first_brace_span(text: str) -> str | None:
"""Return the first balanced ``{...}`` span in ``text`` (string-aware)."""
start = text.find("{")
if start == -1:
return None
depth = 0
in_string = False
escaped = False
for idx in range(start, len(text)):
ch = text[idx]
if in_string:
if escaped:
escaped = False
elif ch == "\\":
escaped = True
elif ch == '"':
in_string = False
continue
if ch == '"':
in_string = True
elif ch == "{":
depth += 1
elif ch == "}":
depth -= 1
if depth == 0:
return text[start : idx + 1]
return None
def _extract_json_object(text: str) -> dict[str, Any] | None:
"""Extract a JSON object from UNTRUSTED model output, or ``None``.
Tolerates the common deviations from "JSON only" (leading apology, trailing
prose, ```json fences) by trying, in order: the whole string, the fenced
block body, then the first balanced ``{...}`` span. Returns ``None`` (never
raises) when nothing parses to a JSON object.
"""
if not isinstance(text, str) or not text.strip():
return None
candidates: list[str] = [_strip_code_fence(text).strip(), text.strip()]
fence = _FENCE_RE.search(text)
if fence:
candidates.append(fence.group("body").strip())
span = _first_brace_span(text)
if span is not None:
candidates.append(span)
for candidate in candidates:
if not candidate:
continue
try:
parsed = json.loads(candidate)
except (json.JSONDecodeError, ValueError):
continue
if isinstance(parsed, dict):
return parsed
return None
def _coerce_mermaid_edits(value: Any) -> list[dict[str, Any]]:
"""Coerce the optional ``mermaid_edits`` into a clean list of edit dicts.
Each kept entry must carry a non-empty ``mermaid`` string; an anchor is kept
when present (either ``macro_id`` or ``anchor``). Anything malformed is
dropped so a garbled edit can never silently corrupt a macro update.
"""
if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
return []
edits: list[dict[str, Any]] = []
for item in value:
if not isinstance(item, Mapping):
continue
mermaid = item.get("mermaid")
if not isinstance(mermaid, str) or not mermaid.strip():
continue
edit: dict[str, Any] = {"mermaid": mermaid.strip()}
macro_id = item.get("macro_id")
anchor = item.get("anchor")
if isinstance(macro_id, str) and macro_id.strip():
edit["macro_id"] = macro_id.strip()
if isinstance(anchor, str) and anchor.strip():
edit["anchor"] = anchor.strip()
edits.append(edit)
return edits
def parse_confluence_reply(text: str) -> dict[str, Any]:
"""Parse the model reply into a validated ``confluence_draft`` dict.
Returns ``{title, body_storage, page_id?, mermaid_edits?}``:
* ``title`` — non-empty page title (required);
* ``body_storage`` — the Confluence storage-format body (required);
* ``page_id`` — included only when the model named an existing page;
* ``mermaid_edits`` — included only when at least one well-formed edit was
supplied.
Raises :class:`ConfluenceDraftError` when the reply is empty, not a JSON
object, or missing a usable ``title`` / ``body_storage`` — a garbled draft
must fail loudly so the node never advances an empty page write.
"""
data = _extract_json_object(text or "")
if data is None:
raise ConfluenceDraftError(
"confluence draft reply was empty or not a JSON object"
)
title = data.get("title")
if not isinstance(title, str) or not title.strip():
raise ConfluenceDraftError("confluence draft is missing a non-empty 'title'")
body = data.get("body_storage")
if not isinstance(body, str) or not body.strip():
raise ConfluenceDraftError(
"confluence draft is missing a non-empty 'body_storage'"
)
draft: dict[str, Any] = {
"title": title.strip(),
"body_storage": body,
}
page_id = data.get("page_id")
if isinstance(page_id, (str, int)) and str(page_id).strip():
page_id_str = str(page_id).strip()
# Confluence content ids are numeric. Reject anything else: a non-numeric
# value (e.g. "123?expand=foo" or "123/child/456") would, if interpolated
# into the REST path, rewrite the authenticated request — a request-path
# injection carrying the service-account credential.
if not re.fullmatch(r"[0-9]+", page_id_str):
raise ConfluenceDraftError(
"confluence draft 'page_id' must be numeric (a Confluence content id)"
)
draft["page_id"] = page_id_str
mermaid_edits = _coerce_mermaid_edits(data.get("mermaid_edits"))
if mermaid_edits:
draft["mermaid_edits"] = mermaid_edits
return draft