"""Prompt-build / reply-parse split for the Confluence-writer node. This module is the pure, model-output-facing half of the Confluence-writer stage — the same ``build_*_prompt`` / ``parse_*_reply`` split :mod:`agent_team.nodes.planner` (``build_plan_prompt`` / ``parse_plan``) and :mod:`agent_team.nodes.clarifier_llm` (``ClaudeClarifier._build_prompt`` / ``_parse``) use, so the node body (:mod:`agent_team.nodes.confluence_writer`) stays a thin LangGraph state-transition wrapper. The Confluence-writer node drafts a **Confluence page update** as DATA (it never writes to a repo, and it never posts to Confluence from the draft node — the write is a later, dry-run-by-default node). Two source flows feed the draft: * **Flow A** — a direct documentation ask: the draft is built from the task description alone (``state['task']``). * **Flow B** — a post-build documentation update: the draft folds in the approved ``plan``, the ``candidate_diff``, and the repo name so the page reflects what the change actually did. On a ``request_changes`` loop-back the prior gate's ``confluence_feedback`` is folded into the prompt so the redraft answers the reviewer's notes (mirrors the planner's ``_format_review_feedback`` loop-back). The LLM binding is an INJECTED seam (``context_provider``), exactly like the planner's, so the prompt assembly stays pure and unit-testable and no live model is bound here. Parsing is DEFENSIVE: the model output is UNTRUSTED. A garbled reply must not silently produce an empty page write, so :func:`parse_confluence_reply` raises :class:`ConfluenceDraftError` when it cannot recover a usable ``{title, body_storage}`` draft, and the node decides what to do with that. """ from __future__ import annotations import json import re from collections.abc import Callable, Mapping, Sequence from typing import Any from agent_team.task_model import PipelineState # Optional context-provider callable (mirrors planner.ContextProvider): () -> str. # When injected, its result is prepended to the Confluence-writer prompt. Default # None = unchanged behavior so existing callers are unaffected. ContextProvider = Callable[[], str] __all__ = [ "ConfluenceDraftError", "build_confluence_prompt", "parse_confluence_reply", ] class ConfluenceDraftError(Exception): """Raised when the model reply cannot be parsed into a usable draft. Distinct from a *gate rejection* (which loops back through the human gate): this signals the draft model output itself was empty/garbled, so the node can fail loudly rather than advance an empty page update toward a write. """ def _task_description(state: PipelineState) -> str: """Pull the task description out of the graph state (mirrors planner.py).""" plan = state.get("plan") or {} if isinstance(plan, dict): desc = plan.get("task") or plan.get("description") if isinstance(desc, str) and desc.strip(): return desc.strip() desc = state.get("task") # type: ignore[call-overload] if isinstance(desc, str) and desc.strip(): return desc.strip() return "" def _repo_name(state: PipelineState) -> str: """Pull the repo name out of the graph state (Flow B), defensively. Looked for under a top-level ``repo`` key and, failing that, inside the plan dict — the same conventional places the clarifier reads ``repo`` from. An absent repo yields an empty string so the prompt section is simply omitted. """ repo = state.get("repo") # type: ignore[call-overload] if isinstance(repo, str) and repo.strip(): return repo.strip() plan = state.get("plan") or {} if isinstance(plan, dict): repo = plan.get("repo") if isinstance(repo, str) and repo.strip(): return repo.strip() return "" def _candidate_diff(state: PipelineState) -> str: """Pull the candidate diff (Flow B) out of the graph state, defensively.""" diff = state.get("candidate_diff") if isinstance(diff, str) and diff.strip(): return diff.strip() return "" def _format_plan(plan: Any) -> str: """Render the approved plan into prompt text (Flow B).""" if not isinstance(plan, Mapping): return "" try: return json.dumps(dict(plan), sort_keys=True, indent=2) except (TypeError, ValueError): return repr(plan) def _truncate(text: str, limit: int) -> str: """Truncate ``text`` to ``limit`` chars with a marker (keep prompts bounded). A candidate diff can be large; the draft only needs enough of it to describe the change, so we cap it rather than blow the context budget. """ if len(text) <= limit: return text return text[:limit] + f"\n... [truncated {len(text) - limit} chars]" def build_confluence_prompt( state: PipelineState, *, context_provider: ContextProvider | None = None, ) -> str: """Build the Claude prompt that drafts a Confluence page update. Pure string assembly over the graph state — no I/O — so the prompt shape is directly unit-testable (mirrors :func:`planner.build_plan_prompt`). Source selection: * **Flow A** (no plan/diff) — the draft is built from ``state['task']``. * **Flow B** (plan and/or candidate diff present) — the draft folds in the approved plan, the candidate diff, and the repo name. On a ``request_changes`` loop-back the prior ``confluence_feedback`` is folded in so the redraft addresses every note (mirrors the planner's review-feedback loop-back). ``context_provider`` is the optional injection seam: when set it is called once and its result prepended (and a raising provider degrades to an empty prefix, never a crash). """ description = _task_description(state) repo = _repo_name(state) diff = _candidate_diff(state) plan = state.get("plan") plan_text = _format_plan(plan) feedback = state.get("confluence_feedback") flow_b = bool(plan_text or diff) sections: list[str] = [] if context_provider is not None: try: ctx = context_provider() except Exception: # noqa: BLE001 - a context provider must never crash the draft ctx = "" if ctx: sections += [ctx, ""] sections += [ "You are the CONFLUENCE-WRITER stage of an agentic SDLC pipeline. " "Inspect the repository (read-only) and draft a Confluence page update " "that documents this work. Emit the draft as DATA only — do NOT edit, " "create, or write any files, and do NOT post to Confluence; a later " "human-gated node performs the actual write.", "", "## Task", description or "(no task description provided)", ] if repo: sections += ["", "## Repository", repo] if flow_b: if plan_text: sections += ["", "## Approved plan (what was decided)", plan_text] if diff: sections += [ "", "## Candidate diff (what the change actually did)", _truncate(diff, 8000), ] sections += [ "", "## Documentation intent", "Update the Confluence page so it reflects the architecture AFTER " "this change: new/removed/modified resources, data flow, and " "configuration. If the page carries an architecture diagram macro, " "describe the mermaid edits needed rather than rewriting unrelated " "content.", ] else: sections += [ "", "## Documentation intent", "Draft the Confluence page update this documentation task asks for.", ] if isinstance(feedback, str) and feedback.strip(): sections += [ "", "## Reviewer feedback on the previous draft (address every point)", feedback.strip(), ] sections += [ "", "## Output format", "Return ONLY a JSON object with keys: " '"title" (string, the page title), ' '"body_storage" (string, the page body in Confluence STORAGE format / ' "XHTML), " '"page_id" (optional string, the id of an existing page to update; omit ' "or null to create), and " '"mermaid_edits" (optional list of objects each with "macro_id" or ' '"anchor" and a "mermaid" string, for architecture-diagram macro ' "updates). Do not include prose outside the JSON. Do NOT use any " "write/edit tools.", ] return "\n".join(sections) # A fenced ```...``` block, if the model wrapped its JSON in Markdown. _FENCE_RE = re.compile( r"```(?:json)?\s*\n?(?P.*?)\n?\s*```", flags=re.DOTALL | re.IGNORECASE, ) def _strip_code_fence(text: str) -> str: """Strip a leading/trailing Markdown code fence if present (mirrors planner).""" fenced = re.match( r"^\s*```(?:json)?\s*\n(?P.*?)\n?\s*```\s*$", text, flags=re.DOTALL | re.IGNORECASE, ) if fenced: return fenced.group("body") return text def _first_brace_span(text: str) -> str | None: """Return the first balanced ``{...}`` span in ``text`` (string-aware).""" start = text.find("{") if start == -1: return None depth = 0 in_string = False escaped = False for idx in range(start, len(text)): ch = text[idx] if in_string: if escaped: escaped = False elif ch == "\\": escaped = True elif ch == '"': in_string = False continue if ch == '"': in_string = True elif ch == "{": depth += 1 elif ch == "}": depth -= 1 if depth == 0: return text[start : idx + 1] return None def _extract_json_object(text: str) -> dict[str, Any] | None: """Extract a JSON object from UNTRUSTED model output, or ``None``. Tolerates the common deviations from "JSON only" (leading apology, trailing prose, ```json fences) by trying, in order: the whole string, the fenced block body, then the first balanced ``{...}`` span. Returns ``None`` (never raises) when nothing parses to a JSON object. """ if not isinstance(text, str) or not text.strip(): return None candidates: list[str] = [_strip_code_fence(text).strip(), text.strip()] fence = _FENCE_RE.search(text) if fence: candidates.append(fence.group("body").strip()) span = _first_brace_span(text) if span is not None: candidates.append(span) for candidate in candidates: if not candidate: continue try: parsed = json.loads(candidate) except (json.JSONDecodeError, ValueError): continue if isinstance(parsed, dict): return parsed return None def _coerce_mermaid_edits(value: Any) -> list[dict[str, Any]]: """Coerce the optional ``mermaid_edits`` into a clean list of edit dicts. Each kept entry must carry a non-empty ``mermaid`` string; an anchor is kept when present (either ``macro_id`` or ``anchor``). Anything malformed is dropped so a garbled edit can never silently corrupt a macro update. """ if not isinstance(value, Sequence) or isinstance(value, (str, bytes)): return [] edits: list[dict[str, Any]] = [] for item in value: if not isinstance(item, Mapping): continue mermaid = item.get("mermaid") if not isinstance(mermaid, str) or not mermaid.strip(): continue edit: dict[str, Any] = {"mermaid": mermaid.strip()} macro_id = item.get("macro_id") anchor = item.get("anchor") if isinstance(macro_id, str) and macro_id.strip(): edit["macro_id"] = macro_id.strip() if isinstance(anchor, str) and anchor.strip(): edit["anchor"] = anchor.strip() edits.append(edit) return edits def parse_confluence_reply(text: str) -> dict[str, Any]: """Parse the model reply into a validated ``confluence_draft`` dict. Returns ``{title, body_storage, page_id?, mermaid_edits?}``: * ``title`` — non-empty page title (required); * ``body_storage`` — the Confluence storage-format body (required); * ``page_id`` — included only when the model named an existing page; * ``mermaid_edits`` — included only when at least one well-formed edit was supplied. Raises :class:`ConfluenceDraftError` when the reply is empty, not a JSON object, or missing a usable ``title`` / ``body_storage`` — a garbled draft must fail loudly so the node never advances an empty page write. """ data = _extract_json_object(text or "") if data is None: raise ConfluenceDraftError( "confluence draft reply was empty or not a JSON object" ) title = data.get("title") if not isinstance(title, str) or not title.strip(): raise ConfluenceDraftError("confluence draft is missing a non-empty 'title'") body = data.get("body_storage") if not isinstance(body, str) or not body.strip(): raise ConfluenceDraftError( "confluence draft is missing a non-empty 'body_storage'" ) draft: dict[str, Any] = { "title": title.strip(), "body_storage": body, } page_id = data.get("page_id") if isinstance(page_id, (str, int)) and str(page_id).strip(): page_id_str = str(page_id).strip() # Confluence content ids are numeric. Reject anything else: a non-numeric # value (e.g. "123?expand=foo" or "123/child/456") would, if interpolated # into the REST path, rewrite the authenticated request — a request-path # injection carrying the service-account credential. if not re.fullmatch(r"[0-9]+", page_id_str): raise ConfluenceDraftError( "confluence draft 'page_id' must be numeric (a Confluence content id)" ) draft["page_id"] = page_id_str mermaid_edits = _coerce_mermaid_edits(data.get("mermaid_edits")) if mermaid_edits: draft["mermaid_edits"] = mermaid_edits return draft