"""Confluence Cloud REST API utilities. Mirrors ``agent/utils/jira.py`` but talks to Confluence Cloud REST v1 (the ``/wiki/rest/api`` base path) with a single service-account (Basic auth over ``email:api_token``). Confluence page/comment bodies are XHTML "storage format," not Atlassian Document Format, so this module has its own tiny storage <-> text converters instead of importing ``agent/utils/adf.py``. """ from __future__ import annotations import base64 import os import re from html import unescape from typing import Any from urllib.parse import quote import httpx from .http import DEFAULT_HTTP_TIMEOUT def _seg(value: str) -> str: """Percent-encode a single untrusted URL path segment.""" return quote(value, safe="") CONFLUENCE_BASE_URL = os.environ.get("CONFLUENCE_BASE_URL", "").rstrip("/") CONFLUENCE_EMAIL = os.environ.get("CONFLUENCE_EMAIL", "") CONFLUENCE_API_TOKEN = os.environ.get("CONFLUENCE_API_TOKEN", "") _PAGE_EXPAND = "body.storage,version,space" def _headers() -> dict[str, str]: token = base64.b64encode(f"{CONFLUENCE_EMAIL}:{CONFLUENCE_API_TOKEN}".encode()).decode() return { "Authorization": f"Basic {token}", "Content-Type": "application/json", "Accept": "application/json", } async def _request( method: str, path: str, *, json: dict[str, Any] | None = None, params: dict[str, Any] | None = None, ) -> dict[str, Any]: """Execute a REST request against the Confluence Cloud API.""" if not (CONFLUENCE_BASE_URL and CONFLUENCE_EMAIL and CONFLUENCE_API_TOKEN): return { "error": "CONFLUENCE_BASE_URL / CONFLUENCE_EMAIL / CONFLUENCE_API_TOKEN are not set" } async with httpx.AsyncClient(timeout=DEFAULT_HTTP_TIMEOUT) as client: try: response = await client.request( method, f"{CONFLUENCE_BASE_URL}/wiki/rest/api{path}", headers=_headers(), json=json, params=params, ) response.raise_for_status() return response.json() if response.content else {} except Exception as e: # noqa: BLE001 return {"error": str(e)} def _page_url(webui_path: str | None) -> str: if not webui_path or not CONFLUENCE_BASE_URL: return "" return f"{CONFLUENCE_BASE_URL}/wiki{webui_path}" def text_to_storage(text: str) -> str: """Wrap blank-line-separated blocks of plain text in ``
`` tags. Minimal converter for agent-authored prose; not a general HTML sanitizer. """ blocks = [b.strip() for b in text.split("\n\n")] escaped = ( b.replace("&", "&").replace("<", "<").replace(">", ">") for b in blocks if b ) return "".join(f"
{b}
" for b in escaped) _TAG_RE = re.compile(r"<[^>]+>") def storage_to_text(xhtml: str) -> str: """Strip XHTML storage-format markup down to plain text.""" if not xhtml: return "" text = _TAG_RE.sub("", xhtml) return unescape(text).strip() def _normalize_page(raw: dict[str, Any]) -> dict[str, Any]: body = raw.get("body", {}) or {} storage = body.get("storage", {}) or {} version = raw.get("version", {}) or {} space = raw.get("space", {}) or {} links = raw.get("_links", {}) or {} return { "id": raw.get("id", ""), "title": raw.get("title", ""), "body": storage_to_text(storage.get("value", "")), "version": version.get("number", 0), "space_key": space.get("key", ""), "url": _page_url(links.get("webui")), } async def get_page(page_id: str) -> dict[str, Any]: """Get a Confluence page by id, with its body normalized to plain text.""" result = await _request("GET", f"/content/{_seg(page_id)}", params={"expand": _PAGE_EXPAND}) if "error" in result: return result return {"page": _normalize_page(result)} async def create_page( space_key: str, title: str, body: str, parent_id: str | None = None, ) -> dict[str, Any]: """Create a new Confluence page in the given space.""" payload: dict[str, Any] = { "type": "page", "title": title, "space": {"key": space_key}, "body": {"storage": {"value": text_to_storage(body), "representation": "storage"}}, } if parent_id is not None: payload["ancestors"] = [{"id": parent_id}] result = await _request("POST", "/content", json=payload) if "error" in result: return result links = result.get("_links", {}) or {} return { "success": bool(result.get("id")), "page": { "id": result.get("id", ""), "title": result.get("title", ""), "url": _page_url(links.get("webui")), }, } async def update_page( page_id: str, title: str | None = None, body: str | None = None, ) -> dict[str, Any]: """Update an existing Confluence page. Confluence requires the next version number on every update, so this first reads the current page to learn ``version.number`` and the existing title (title is a required field on the PUT even when unchanged). """ current = await get_page(page_id) if "error" in current: return current page = current["page"] new_title = title if title is not None else page["title"] payload: dict[str, Any] = { "type": "page", "title": new_title, "version": {"number": page["version"] + 1}, } if body is not None: payload["body"] = {"storage": {"value": text_to_storage(body), "representation": "storage"}} result = await _request("PUT", f"/content/{_seg(page_id)}", json=payload) if "error" in result: return result links = result.get("_links", {}) or {} return { "success": bool(result.get("id")), "page": { "id": result.get("id", ""), "title": result.get("title", ""), "url": _page_url(links.get("webui")), }, } async def add_comment(page_id: str, body: str) -> dict[str, Any]: """Add a comment to a Confluence page.""" payload = { "type": "comment", "container": {"id": page_id, "type": "page"}, "body": {"storage": {"value": text_to_storage(body), "representation": "storage"}}, } result = await _request("POST", "/content", json=payload) if "error" in result: return result return {"success": bool(result.get("id")), "id": result.get("id", "")} def _normalize_search_result(raw: dict[str, Any]) -> dict[str, Any]: links = raw.get("_links", {}) or {} return { "id": raw.get("id", ""), "title": raw.get("title", ""), "type": raw.get("type", ""), "url": _page_url(links.get("webui")), } async def search(cql: str) -> dict[str, Any]: """Search Confluence content using CQL.""" result = await _request("GET", "/content/search", params={"cql": cql}) if "error" in result: return result results = result.get("results", []) return {"results": [_normalize_search_result(r) for r in results]} _COMMENT_EXPAND = "body.storage,history.createdBy,ancestors,space" def _normalize_comment(raw: dict[str, Any]) -> dict[str, Any]: body = (raw.get("body") or {}).get("storage") or {} author = ((raw.get("history") or {}).get("createdBy")) or {} ancestors = raw.get("ancestors") or [] space = raw.get("space") or {} # A comment's container page is its nearest ancestor. page_id = ancestors[-1].get("id", "") if ancestors else "" return { "id": raw.get("id", ""), "body": storage_to_text(body.get("value", "")), "author": { "account_id": author.get("accountId"), "name": author.get("displayName"), "email": author.get("email"), }, "page_id": page_id, "space_key": space.get("key", ""), } async def get_comment(comment_id: str) -> dict[str, Any]: """Fetch a Confluence comment by id — the authoritative record for a webhook. Connect webhook bodies are only a pointer; the comment's real author, text, and container are read here via the Basic-auth service account. """ result = await _request( "GET", f"/content/{_seg(comment_id)}", params={"expand": _COMMENT_EXPAND} ) if "error" in result: return result return {"comment": _normalize_comment(result)} async def get_user_email(account_id: str) -> str | None: """Look up a Confluence user's email by accountId (webhooks carry accountId).""" result = await _request("GET", "/user", params={"accountId": account_id}) if "error" in result: return None return result.get("email")