Add a Confluence documentation lane to the Plane-2 pipeline, flag-gated behind AGENT_TEAM_CONFLUENCE_ENABLED (default off; daemon behavior unchanged when off). - confluence/client.py: OAuth 2LO + Basic REST client, dry-run-default writes - confluence/mermaid.py: vendored ADF-only Mermaid editor (macro-count + revert-diff guards, dry-run default) - nodes/confluence_writer.py(+_llm): conf_draft -> conf_gate -> conf_write, both direct (task_kind=confluence) and post-build documentation flows - task_model/graph/coordinator: new phases, state channels, route_after_intake, CONFLUENCE_APPROVAL_KIND gate delivery, task_kind forwarding - db schema v5: widen pending_questions kind CHECK (atomic rebuild) - tests for client, mermaid, writer node, ledger v5, coordinator gate, e2e
384 lines
14 KiB
Python
384 lines
14 KiB
Python
"""Prompt-build / reply-parse split for the Confluence-writer node.
|
|
|
|
This module is the pure, model-output-facing half of the Confluence-writer
|
|
stage — the same ``build_*_prompt`` / ``parse_*_reply`` split
|
|
:mod:`agent_team.nodes.planner` (``build_plan_prompt`` / ``parse_plan``) and
|
|
:mod:`agent_team.nodes.clarifier_llm` (``ClaudeClarifier._build_prompt`` /
|
|
``_parse``) use, so the node body (:mod:`agent_team.nodes.confluence_writer`)
|
|
stays a thin LangGraph state-transition wrapper.
|
|
|
|
The Confluence-writer node drafts a **Confluence page update** as DATA (it never
|
|
writes to a repo, and it never posts to Confluence from the draft node — the
|
|
write is a later, dry-run-by-default node). Two source flows feed the draft:
|
|
|
|
* **Flow A** — a direct documentation ask: the draft is built from the task
|
|
description alone (``state['task']``).
|
|
* **Flow B** — a post-build documentation update: the draft folds in the
|
|
approved ``plan``, the ``candidate_diff``, and the repo name so the page
|
|
reflects what the change actually did.
|
|
|
|
On a ``request_changes`` loop-back the prior gate's ``confluence_feedback`` is
|
|
folded into the prompt so the redraft answers the reviewer's notes (mirrors the
|
|
planner's ``_format_review_feedback`` loop-back).
|
|
|
|
The LLM binding is an INJECTED seam (``context_provider``), exactly like the
|
|
planner's, so the prompt assembly stays pure and unit-testable and no live model
|
|
is bound here.
|
|
|
|
Parsing is DEFENSIVE: the model output is UNTRUSTED. A garbled reply must not
|
|
silently produce an empty page write, so :func:`parse_confluence_reply` raises
|
|
:class:`ConfluenceDraftError` when it cannot recover a usable
|
|
``{title, body_storage}`` draft, and the node decides what to do with that.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from collections.abc import Callable, Mapping, Sequence
|
|
from typing import Any
|
|
|
|
from agent_team.task_model import PipelineState
|
|
|
|
# Optional context-provider callable (mirrors planner.ContextProvider): () -> str.
|
|
# When injected, its result is prepended to the Confluence-writer prompt. Default
|
|
# None = unchanged behavior so existing callers are unaffected.
|
|
ContextProvider = Callable[[], str]
|
|
|
|
__all__ = [
|
|
"ConfluenceDraftError",
|
|
"build_confluence_prompt",
|
|
"parse_confluence_reply",
|
|
]
|
|
|
|
|
|
class ConfluenceDraftError(Exception):
|
|
"""Raised when the model reply cannot be parsed into a usable draft.
|
|
|
|
Distinct from a *gate rejection* (which loops back through the human gate):
|
|
this signals the draft model output itself was empty/garbled, so the node
|
|
can fail loudly rather than advance an empty page update toward a write.
|
|
"""
|
|
|
|
|
|
def _task_description(state: PipelineState) -> str:
|
|
"""Pull the task description out of the graph state (mirrors planner.py)."""
|
|
plan = state.get("plan") or {}
|
|
if isinstance(plan, dict):
|
|
desc = plan.get("task") or plan.get("description")
|
|
if isinstance(desc, str) and desc.strip():
|
|
return desc.strip()
|
|
desc = state.get("task") # type: ignore[call-overload]
|
|
if isinstance(desc, str) and desc.strip():
|
|
return desc.strip()
|
|
return ""
|
|
|
|
|
|
def _repo_name(state: PipelineState) -> str:
|
|
"""Pull the repo name out of the graph state (Flow B), defensively.
|
|
|
|
Looked for under a top-level ``repo`` key and, failing that, inside the plan
|
|
dict — the same conventional places the clarifier reads ``repo`` from. An
|
|
absent repo yields an empty string so the prompt section is simply omitted.
|
|
"""
|
|
repo = state.get("repo") # type: ignore[call-overload]
|
|
if isinstance(repo, str) and repo.strip():
|
|
return repo.strip()
|
|
plan = state.get("plan") or {}
|
|
if isinstance(plan, dict):
|
|
repo = plan.get("repo")
|
|
if isinstance(repo, str) and repo.strip():
|
|
return repo.strip()
|
|
return ""
|
|
|
|
|
|
def _candidate_diff(state: PipelineState) -> str:
|
|
"""Pull the candidate diff (Flow B) out of the graph state, defensively."""
|
|
diff = state.get("candidate_diff")
|
|
if isinstance(diff, str) and diff.strip():
|
|
return diff.strip()
|
|
return ""
|
|
|
|
|
|
def _format_plan(plan: Any) -> str:
|
|
"""Render the approved plan into prompt text (Flow B)."""
|
|
if not isinstance(plan, Mapping):
|
|
return ""
|
|
try:
|
|
return json.dumps(dict(plan), sort_keys=True, indent=2)
|
|
except (TypeError, ValueError):
|
|
return repr(plan)
|
|
|
|
|
|
def _truncate(text: str, limit: int) -> str:
|
|
"""Truncate ``text`` to ``limit`` chars with a marker (keep prompts bounded).
|
|
|
|
A candidate diff can be large; the draft only needs enough of it to describe
|
|
the change, so we cap it rather than blow the context budget.
|
|
"""
|
|
if len(text) <= limit:
|
|
return text
|
|
return text[:limit] + f"\n... [truncated {len(text) - limit} chars]"
|
|
|
|
|
|
def build_confluence_prompt(
|
|
state: PipelineState,
|
|
*,
|
|
context_provider: ContextProvider | None = None,
|
|
) -> str:
|
|
"""Build the Claude prompt that drafts a Confluence page update.
|
|
|
|
Pure string assembly over the graph state — no I/O — so the prompt shape is
|
|
directly unit-testable (mirrors :func:`planner.build_plan_prompt`).
|
|
|
|
Source selection:
|
|
|
|
* **Flow A** (no plan/diff) — the draft is built from ``state['task']``.
|
|
* **Flow B** (plan and/or candidate diff present) — the draft folds in the
|
|
approved plan, the candidate diff, and the repo name.
|
|
|
|
On a ``request_changes`` loop-back the prior ``confluence_feedback`` is
|
|
folded in so the redraft addresses every note (mirrors the planner's
|
|
review-feedback loop-back). ``context_provider`` is the optional injection
|
|
seam: when set it is called once and its result prepended (and a raising
|
|
provider degrades to an empty prefix, never a crash).
|
|
"""
|
|
description = _task_description(state)
|
|
repo = _repo_name(state)
|
|
diff = _candidate_diff(state)
|
|
plan = state.get("plan")
|
|
plan_text = _format_plan(plan)
|
|
feedback = state.get("confluence_feedback")
|
|
flow_b = bool(plan_text or diff)
|
|
|
|
sections: list[str] = []
|
|
|
|
if context_provider is not None:
|
|
try:
|
|
ctx = context_provider()
|
|
except Exception: # noqa: BLE001 - a context provider must never crash the draft
|
|
ctx = ""
|
|
if ctx:
|
|
sections += [ctx, ""]
|
|
|
|
sections += [
|
|
"You are the CONFLUENCE-WRITER stage of an agentic SDLC pipeline. "
|
|
"Inspect the repository (read-only) and draft a Confluence page update "
|
|
"that documents this work. Emit the draft as DATA only — do NOT edit, "
|
|
"create, or write any files, and do NOT post to Confluence; a later "
|
|
"human-gated node performs the actual write.",
|
|
"",
|
|
"## Task",
|
|
description or "(no task description provided)",
|
|
]
|
|
|
|
if repo:
|
|
sections += ["", "## Repository", repo]
|
|
|
|
if flow_b:
|
|
if plan_text:
|
|
sections += ["", "## Approved plan (what was decided)", plan_text]
|
|
if diff:
|
|
sections += [
|
|
"",
|
|
"## Candidate diff (what the change actually did)",
|
|
_truncate(diff, 8000),
|
|
]
|
|
sections += [
|
|
"",
|
|
"## Documentation intent",
|
|
"Update the Confluence page so it reflects the architecture AFTER "
|
|
"this change: new/removed/modified resources, data flow, and "
|
|
"configuration. If the page carries an architecture diagram macro, "
|
|
"describe the mermaid edits needed rather than rewriting unrelated "
|
|
"content.",
|
|
]
|
|
else:
|
|
sections += [
|
|
"",
|
|
"## Documentation intent",
|
|
"Draft the Confluence page update this documentation task asks for.",
|
|
]
|
|
|
|
if isinstance(feedback, str) and feedback.strip():
|
|
sections += [
|
|
"",
|
|
"## Reviewer feedback on the previous draft (address every point)",
|
|
feedback.strip(),
|
|
]
|
|
|
|
sections += [
|
|
"",
|
|
"## Output format",
|
|
"Return ONLY a JSON object with keys: "
|
|
'"title" (string, the page title), '
|
|
'"body_storage" (string, the page body in Confluence STORAGE format / '
|
|
"XHTML), "
|
|
'"page_id" (optional string, the id of an existing page to update; omit '
|
|
"or null to create), and "
|
|
'"mermaid_edits" (optional list of objects each with "macro_id" or '
|
|
'"anchor" and a "mermaid" string, for architecture-diagram macro '
|
|
"updates). Do not include prose outside the JSON. Do NOT use any "
|
|
"write/edit tools.",
|
|
]
|
|
return "\n".join(sections)
|
|
|
|
|
|
# A fenced ```...``` block, if the model wrapped its JSON in Markdown.
|
|
_FENCE_RE = re.compile(
|
|
r"```(?:json)?\s*\n?(?P<body>.*?)\n?\s*```",
|
|
flags=re.DOTALL | re.IGNORECASE,
|
|
)
|
|
|
|
|
|
def _strip_code_fence(text: str) -> str:
|
|
"""Strip a leading/trailing Markdown code fence if present (mirrors planner)."""
|
|
fenced = re.match(
|
|
r"^\s*```(?:json)?\s*\n(?P<body>.*?)\n?\s*```\s*$",
|
|
text,
|
|
flags=re.DOTALL | re.IGNORECASE,
|
|
)
|
|
if fenced:
|
|
return fenced.group("body")
|
|
return text
|
|
|
|
|
|
def _first_brace_span(text: str) -> str | None:
|
|
"""Return the first balanced ``{...}`` span in ``text`` (string-aware)."""
|
|
start = text.find("{")
|
|
if start == -1:
|
|
return None
|
|
depth = 0
|
|
in_string = False
|
|
escaped = False
|
|
for idx in range(start, len(text)):
|
|
ch = text[idx]
|
|
if in_string:
|
|
if escaped:
|
|
escaped = False
|
|
elif ch == "\\":
|
|
escaped = True
|
|
elif ch == '"':
|
|
in_string = False
|
|
continue
|
|
if ch == '"':
|
|
in_string = True
|
|
elif ch == "{":
|
|
depth += 1
|
|
elif ch == "}":
|
|
depth -= 1
|
|
if depth == 0:
|
|
return text[start : idx + 1]
|
|
return None
|
|
|
|
|
|
def _extract_json_object(text: str) -> dict[str, Any] | None:
|
|
"""Extract a JSON object from UNTRUSTED model output, or ``None``.
|
|
|
|
Tolerates the common deviations from "JSON only" (leading apology, trailing
|
|
prose, ```json fences) by trying, in order: the whole string, the fenced
|
|
block body, then the first balanced ``{...}`` span. Returns ``None`` (never
|
|
raises) when nothing parses to a JSON object.
|
|
"""
|
|
if not isinstance(text, str) or not text.strip():
|
|
return None
|
|
candidates: list[str] = [_strip_code_fence(text).strip(), text.strip()]
|
|
fence = _FENCE_RE.search(text)
|
|
if fence:
|
|
candidates.append(fence.group("body").strip())
|
|
span = _first_brace_span(text)
|
|
if span is not None:
|
|
candidates.append(span)
|
|
for candidate in candidates:
|
|
if not candidate:
|
|
continue
|
|
try:
|
|
parsed = json.loads(candidate)
|
|
except (json.JSONDecodeError, ValueError):
|
|
continue
|
|
if isinstance(parsed, dict):
|
|
return parsed
|
|
return None
|
|
|
|
|
|
def _coerce_mermaid_edits(value: Any) -> list[dict[str, Any]]:
|
|
"""Coerce the optional ``mermaid_edits`` into a clean list of edit dicts.
|
|
|
|
Each kept entry must carry a non-empty ``mermaid`` string; an anchor is kept
|
|
when present (either ``macro_id`` or ``anchor``). Anything malformed is
|
|
dropped so a garbled edit can never silently corrupt a macro update.
|
|
"""
|
|
if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
|
|
return []
|
|
edits: list[dict[str, Any]] = []
|
|
for item in value:
|
|
if not isinstance(item, Mapping):
|
|
continue
|
|
mermaid = item.get("mermaid")
|
|
if not isinstance(mermaid, str) or not mermaid.strip():
|
|
continue
|
|
edit: dict[str, Any] = {"mermaid": mermaid.strip()}
|
|
macro_id = item.get("macro_id")
|
|
anchor = item.get("anchor")
|
|
if isinstance(macro_id, str) and macro_id.strip():
|
|
edit["macro_id"] = macro_id.strip()
|
|
if isinstance(anchor, str) and anchor.strip():
|
|
edit["anchor"] = anchor.strip()
|
|
edits.append(edit)
|
|
return edits
|
|
|
|
|
|
def parse_confluence_reply(text: str) -> dict[str, Any]:
|
|
"""Parse the model reply into a validated ``confluence_draft`` dict.
|
|
|
|
Returns ``{title, body_storage, page_id?, mermaid_edits?}``:
|
|
|
|
* ``title`` — non-empty page title (required);
|
|
* ``body_storage`` — the Confluence storage-format body (required);
|
|
* ``page_id`` — included only when the model named an existing page;
|
|
* ``mermaid_edits`` — included only when at least one well-formed edit was
|
|
supplied.
|
|
|
|
Raises :class:`ConfluenceDraftError` when the reply is empty, not a JSON
|
|
object, or missing a usable ``title`` / ``body_storage`` — a garbled draft
|
|
must fail loudly so the node never advances an empty page write.
|
|
"""
|
|
data = _extract_json_object(text or "")
|
|
if data is None:
|
|
raise ConfluenceDraftError(
|
|
"confluence draft reply was empty or not a JSON object"
|
|
)
|
|
|
|
title = data.get("title")
|
|
if not isinstance(title, str) or not title.strip():
|
|
raise ConfluenceDraftError("confluence draft is missing a non-empty 'title'")
|
|
|
|
body = data.get("body_storage")
|
|
if not isinstance(body, str) or not body.strip():
|
|
raise ConfluenceDraftError(
|
|
"confluence draft is missing a non-empty 'body_storage'"
|
|
)
|
|
|
|
draft: dict[str, Any] = {
|
|
"title": title.strip(),
|
|
"body_storage": body,
|
|
}
|
|
|
|
page_id = data.get("page_id")
|
|
if isinstance(page_id, (str, int)) and str(page_id).strip():
|
|
page_id_str = str(page_id).strip()
|
|
# Confluence content ids are numeric. Reject anything else: a non-numeric
|
|
# value (e.g. "123?expand=foo" or "123/child/456") would, if interpolated
|
|
# into the REST path, rewrite the authenticated request — a request-path
|
|
# injection carrying the service-account credential.
|
|
if not re.fullmatch(r"[0-9]+", page_id_str):
|
|
raise ConfluenceDraftError(
|
|
"confluence draft 'page_id' must be numeric (a Confluence content id)"
|
|
)
|
|
draft["page_id"] = page_id_str
|
|
|
|
mermaid_edits = _coerce_mermaid_edits(data.get("mermaid_edits"))
|
|
if mermaid_edits:
|
|
draft["mermaid_edits"] = mermaid_edits
|
|
|
|
return draft
|