"""Confluence REST v2 client with a dry-run-first page updater. This module ports the auth seam from the security-review bash reference (``checkers/confluence-doc.sh`` ``conf_api_init`` / ``conf_get``, lines ~210-304) into a typed Python client suited to the durable pipeline: Auth (OAuth wins when present), mirroring the bash ``conf_api_init`` precedence: A. **2LO client-credentials (OAuth).** When ``CONFLUENCE_OAUTH_CLIENT_ID`` and ``CONFLUENCE_OAUTH_CLIENT_SECRET`` are set, POST a ``grant_type=client_credentials`` request to ``CONFLUENCE_OAUTH_TOKEN_URL`` (default ``https://auth.atlassian.com/oauth/token``) for a Bearer token, then resolve the cloudId from ``CONFLUENCE_CLOUD_ID`` or the OAuth-native ``accessible-resources`` endpoint. Base becomes ``https://api.atlassian.com/ex/confluence/``. B. **Basic auth.** When ``CONFLUENCE_BASE_URL`` + ``CONFLUENCE_EMAIL`` + ``CONFLUENCE_API_TOKEN`` are set, use HTTP Basic against the configured base. As in the bash reference, a missing/incomplete cred set or a failed token / cloudId resolution surfaces cleanly (:class:`ConfluenceError`) so the caller can SKIP rather than raise a false alarm — mirroring ``conf_api_init`` returning non-zero. I/O seam: HTTP is the injected :class:`HttpClient` protocol so tests pass an in-memory fake and no live network is touched in any code path the tests hit. The default :class:`UrllibHttpClient` is stdlib-only (``urllib``), matching ``agent_team.transport.github_adapter`` — no third-party dependency. Secrets (``CONFLUENCE_*``) are read from ``os.environ`` at CALL time via :func:`init_auth`, never captured at import or stored long-lived on the module. Dry-run-first updater: :meth:`ConfluenceClient.update_page` defaults to ``apply=False``. In dry-run it returns a :class:`PlannedPageUpdate` (target id, ``new_version = current + 1``, and a unified-diff body delta) and performs NO network write. Only ``apply=True`` issues the PUT. """ from __future__ import annotations import base64 import difflib import json import os import re from collections import Counter from dataclasses import dataclass, field from typing import Any, Protocol from urllib import error as _urlerror from urllib import parse as _urlparse from urllib import request as _urlrequest __all__ = [ "DEFAULT_OAUTH_TOKEN_URL", "ATLASSIAN_API_BASE", "ACCESSIBLE_RESOURCES_URL", "ConfluenceAuth", "ConfluenceClient", "ConfluenceError", "HttpClient", "PlannedPageUpdate", "UrllibHttpClient", "body_diff", "count_storage_macros", "init_auth", "storage_macro_signature", ] # Confluence storage-format macros (Mermaid diagrams, info panels, etc.) are # serialised as ```` / ```` elements whose # payload does NOT survive a wholesale body replacement. Counting them lets a # writer refuse a storage update that would silently drop diagram macros — the # exact failure that erased every diagram on page 1540098. Case-insensitive. _STORAGE_MACRO_RE = re.compile( r"]*`` matches both self-closing and paired opening tags. _STORAGE_MACRO_TAG_RE = re.compile( r"]*)>", re.IGNORECASE ) _AC_NAME_RE = re.compile(r'ac:name\s*=\s*"([^"]*)"', re.IGNORECASE) # Default 2LO token endpoint (overridable via CONFLUENCE_OAUTH_TOKEN_URL). DEFAULT_OAUTH_TOKEN_URL = "https://auth.atlassian.com/oauth/token" # OAuth-native gateway base for per-cloud Confluence access. ATLASSIAN_API_BASE = "https://api.atlassian.com/ex/confluence" # OAuth-native resource enumeration used to resolve the cloudId. ACCESSIBLE_RESOURCES_URL = "https://api.atlassian.com/oauth/token/accessible-resources" class ConfluenceError(RuntimeError): """Raised when auth resolution or a REST call fails. Carries an HTTP-ish ``status`` (``0`` for pre-flight/config failures such as missing creds or an unresolvable cloudId) and a (truncated) ``body``. A caller that wants the bash ``conf_api_init``-style "skip, no false alarm" behaviour can catch this and skip rather than propagate. """ def __init__(self, status: int, body: str) -> None: self.status = status self.body = body super().__init__(f"Confluence error {status}: {body[:200]}") class HttpClient(Protocol): """Injected HTTP seam returning ``(status, body)`` for each verb. ``body`` is the parsed JSON (a ``dict``) when the response carries JSON, else the raw ``bytes``. Keeping the surface narrow (``get``/``post``/``put``) means the client has no hard HTTP dependency and tests inject a pure in-memory fake — no live network on any tested path. """ def get( self, url: str, *, headers: dict[str, str] ) -> tuple[int, dict[str, Any] | bytes]: ... def post( self, url: str, *, headers: dict[str, str], data: bytes, ) -> tuple[int, dict[str, Any] | bytes]: ... def put( self, url: str, *, headers: dict[str, str], data: bytes, ) -> tuple[int, dict[str, Any] | bytes]: ... # --------------------------------------------------------------------------- # Default stdlib-only HTTP impl (no third-party dependency at import time). # --------------------------------------------------------------------------- def _parse_body(raw: bytes, content_type: str) -> dict[str, Any] | bytes: """Parse a response body to a ``dict`` when it is JSON, else return bytes.""" if not raw: return {} if "application/json" in (content_type or "").lower(): try: return json.loads(raw.decode("utf-8")) except (ValueError, UnicodeDecodeError): return raw return raw class UrllibHttpClient: """Stdlib-only :class:`HttpClient` (``urllib``), built lazily. Used only when no client is injected and an actual request is made; tests never reach this path because they inject a fake. Matches the ``github_adapter`` pattern: urls are built from fixed ``https`` Atlassian bases, so there is no SSRF/``file://`` surface. """ def _request( self, url: str, *, method: str, headers: dict[str, str], data: bytes | None, ) -> tuple[int, dict[str, Any] | bytes]: request = _urlrequest.Request(url, data=data, method=method) for key, value in headers.items(): request.add_header(key, value) try: with _urlrequest.urlopen(request) as response: # noqa: S310 (trusted atlassian host); nosemgrep status = response.getcode() raw = response.read() content_type = response.headers.get("Content-Type", "") except _urlerror.HTTPError as exc: # pragma: no cover - network path raw = exc.read() content_type = exc.headers.get("Content-Type", "") if exc.headers else "" return exc.code, _parse_body(raw, content_type) return status, _parse_body(raw, content_type) def get( self, url: str, *, headers: dict[str, str] ) -> tuple[int, dict[str, Any] | bytes]: return self._request(url, method="GET", headers=headers, data=None) def post( self, url: str, *, headers: dict[str, str], data: bytes ) -> tuple[int, dict[str, Any] | bytes]: return self._request(url, method="POST", headers=headers, data=data) def put( self, url: str, *, headers: dict[str, str], data: bytes ) -> tuple[int, dict[str, Any] | bytes]: return self._request(url, method="PUT", headers=headers, data=data) # --------------------------------------------------------------------------- # Auth resolution (pure-ish: env in, resolved config out; one HTTP seam for OAuth). # --------------------------------------------------------------------------- @dataclass(frozen=True) class ConfluenceAuth: """Resolved auth context: the base URL plus per-request headers. Exactly one of the two modes is materialised (``"oauth"`` or ``"basic"``), mirroring ``conf_api_init`` resolving ONE mode. ``base`` already includes the OAuth ``/ex/confluence/`` segment when relevant; both modes then share the ``/wiki/api/v2/...`` path suffix exactly as ``conf_get`` does. """ mode: str base: str _headers: dict[str, str] = field(default_factory=dict) def headers(self, *, extra: dict[str, str] | None = None) -> dict[str, str]: """Per-request headers (auth + accept), merged with ``extra``.""" merged = dict(self._headers) if extra: merged.update(extra) return merged def _resolve_cloud_id( http: HttpClient, *, bearer: str, configured_cloud_id: str | None, site_url: str | None, ) -> str: """Resolve the cloudId, preferring config then accessible-resources. Mirrors the bash fallback: use ``CONFLUENCE_CLOUD_ID`` when set, else query the OAuth-native ``accessible-resources`` endpoint and prefer the resource whose ``url`` matches the configured site, else the first one. """ if configured_cloud_id: return configured_cloud_id status, body = http.get( ACCESSIBLE_RESOURCES_URL, headers={"Authorization": f"Bearer {bearer}", "Accept": "application/json"}, ) if not (200 <= status < 300) or not isinstance(body, list): # bash treats this as "could not resolve cloudId -> skip" (status 0). raise ConfluenceError( 0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)" ) chosen: str | None = None if site_url: for resource in body: if isinstance(resource, dict) and resource.get("url") == site_url: chosen = resource.get("id") break if not chosen and not site_url: # No site_url to disambiguate: auto-resolve ONLY when there is exactly # one accessible resource. Silently picking the first of several could # target the WRONG Confluence site (a confused-deputy / wrong-blast-radius # hazard) — require an explicit CONFLUENCE_CLOUD_ID instead. candidates = [r["id"] for r in body if isinstance(r, dict) and r.get("id")] if len(candidates) == 1: chosen = candidates[0] elif len(candidates) > 1: raise ConfluenceError( 0, "OAuth: multiple accessible Confluence sites; set " "CONFLUENCE_CLOUD_ID (or CONFLUENCE_BASE_URL) to disambiguate", ) if not chosen: raise ConfluenceError( 0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)" ) return chosen def init_auth(http: HttpClient, *, env: dict[str, str] | None = None) -> ConfluenceAuth: """Resolve ONE auth mode from the environment at call time (OAuth wins). Reads ``CONFLUENCE_*`` from ``env`` (default ``os.environ``) only when invoked, never at import. Raises :class:`ConfluenceError` (``status=0``) on a missing/incomplete cred set or a failed token / cloudId resolution — the ``conf_api_init`` "return non-zero -> caller skips" contract, so a missing cred path is cleanly detectable. Args: http: HTTP seam (used only for the OAuth token + cloudId calls). env: Environment mapping to read creds from; defaults to ``os.environ``. Returns: A :class:`ConfluenceAuth` with the resolved base + auth headers. """ environ = os.environ if env is None else env client_id = environ.get("CONFLUENCE_OAUTH_CLIENT_ID") client_secret = environ.get("CONFLUENCE_OAUTH_CLIENT_SECRET") site_url = environ.get("CONFLUENCE_BASE_URL") # Mode A: 2LO client-credentials (OAuth wins when its creds are present). if client_id and client_secret: token_url = environ.get("CONFLUENCE_OAUTH_TOKEN_URL") or DEFAULT_OAUTH_TOKEN_URL form = _urlparse.urlencode( { "client_id": client_id, "client_secret": client_secret, "grant_type": "client_credentials", } ).encode("utf-8") status, body = http.post( token_url, headers={"Content-Type": "application/x-www-form-urlencoded"}, data=form, ) bearer = body.get("access_token") if isinstance(body, dict) else None if not (200 <= status < 300) or not bearer: # bash: "OAuth: token request failed — skipping API (no false alarm)". raise ConfluenceError(0, "OAuth: token request failed") cloud_id = _resolve_cloud_id( http, bearer=bearer, configured_cloud_id=environ.get("CONFLUENCE_CLOUD_ID"), site_url=site_url, ) return ConfluenceAuth( mode="oauth", base=f"{ATLASSIAN_API_BASE}/{cloud_id}", _headers={ "Authorization": f"Bearer {bearer}", "Accept": "application/json", }, ) # Mode B: Basic auth (email + API token against the configured base). email = environ.get("CONFLUENCE_EMAIL") api_token = environ.get("CONFLUENCE_API_TOKEN") if site_url and email and api_token: raw = f"{email}:{api_token}".encode("utf-8") encoded = base64.b64encode(raw).decode("ascii") return ConfluenceAuth( mode="basic", base=site_url.rstrip("/"), _headers={ "Authorization": f"Basic {encoded}", "Accept": "application/json", }, ) # Neither mode fully configured -> caller skips (no false alarm). raise ConfluenceError(0, "no Confluence credentials configured (OAuth or Basic)") # --------------------------------------------------------------------------- # Pure planning logic (no I/O) — kept separate so it is unit-trivial. # --------------------------------------------------------------------------- def body_diff(old_body: str, new_body: str, *, page_id: str) -> str: """Render a unified-diff of a page's storage body (old -> new). Pure string logic with no I/O, so the dry-run plan is testable without any HTTP. Empty string when the bodies are identical. """ old_lines = (old_body or "").splitlines(keepends=True) new_lines = (new_body or "").splitlines(keepends=True) diff = difflib.unified_diff( old_lines, new_lines, fromfile=f"page/{page_id}@current", tofile=f"page/{page_id}@planned", ) return "".join(diff) @dataclass(frozen=True) class PlannedPageUpdate: """A dry-run page update: the change that WOULD be applied, no PUT issued. Returned by :meth:`ConfluenceClient.update_page` when ``apply=False`` (the default). ``applied`` is ``False`` here; the same dataclass is returned with ``applied=True`` after a real PUT so callers get a uniform shape. """ page_id: str title: str current_version: int new_version: int body_storage: str body_delta: str applied: bool = False # --------------------------------------------------------------------------- # The client (thin I/O wrapper over the seam + the pure planning logic). # --------------------------------------------------------------------------- class ConfluenceClient: """Read pages and plan/apply page updates over the injected HTTP seam. Auth is resolved lazily on first use (or eagerly if a :class:`ConfluenceAuth` is injected), reading ``CONFLUENCE_*`` from the environment at call time. The HTTP seam is injected so tests stay hermetic. """ def __init__( self, *, http: HttpClient | None = None, auth: ConfluenceAuth | None = None, env: dict[str, str] | None = None, ) -> None: """Construct the client. Args: http: Injected HTTP seam. Defaults to a stdlib-only :class:`UrllibHttpClient` built lazily (tests inject a fake). auth: Pre-resolved auth context. When omitted, auth is resolved on first use from the environment via :func:`init_auth`. env: Environment mapping for cred resolution; defaults to ``os.environ`` (read at call time, never at import). """ self._http: HttpClient = http or UrllibHttpClient() self._auth = auth self._env = env def _ensure_auth(self) -> ConfluenceAuth: """Resolve auth on first use; raises :class:`ConfluenceError` if missing.""" if self._auth is None: self._auth = init_auth(self._http, env=self._env) return self._auth @staticmethod def _as_dict(body: dict[str, Any] | bytes) -> dict[str, Any]: """Coerce a response body to a dict or raise a clean parse error.""" if isinstance(body, dict): return body raise ConfluenceError(0, "expected JSON object response, got non-JSON body") def get_page(self, page_id: str) -> dict[str, Any]: """GET ``/wiki/api/v2/pages/{id}?body-format=storage``. Returns the parsed page object (including ``version.number`` and ``body.storage.value``). Raises :class:`ConfluenceError` on a non-2xx response, carrying the status so the caller can branch (e.g. 404 = gone). """ auth = self._ensure_auth() # Defense-in-depth: percent-encode the id segment so a malformed id can # never rewrite the request path (request-path injection). id_segment = _urlparse.quote(str(page_id), safe="") url = f"{auth.base}/wiki/api/v2/pages/{id_segment}?body-format=storage" status, body = self._http.get(url, headers=auth.headers()) if not (200 <= status < 300): raise ConfluenceError(status, _stringify(body)) return self._as_dict(body) def page_has_macros(self, page_id: str) -> bool: """Whether the page's CURRENT storage body carries Confluence macros. Reads the page (storage format) and counts ```` / ```` elements (e.g. Mermaid diagram extensions). The writer node uses this to refuse a wholesale storage overwrite that would drop diagram macros. Raises :class:`ConfluenceError` if the page cannot be read — the caller treats an unreadable page as "assume macros" (fail closed) rather than overwriting blindly. """ page = self.get_page(str(page_id)) return count_storage_macros(_extract_storage_body(page)) > 0 def update_page( self, page_id: str, title: str, body_storage: str, version_number: int, *, apply: bool = False, ) -> PlannedPageUpdate: """Plan (default) or apply a page update via ``/wiki/api/v2/pages/{id}``. CRITICAL — dry-run by default. With ``apply=False`` (the default) NO network write happens: this returns a :class:`PlannedPageUpdate` describing the change (target id, ``new_version = version_number + 1``, and a unified-diff ``body_delta`` against the page's current storage body). Only ``apply=True`` issues the PUT. Args: page_id: Target page id. title: New page title (Confluence requires title on update). body_storage: New body in Confluence ``storage`` representation. version_number: The page's CURRENT version number; the PUT/plan uses ``version_number + 1`` as the new version (Confluence's optimistic-concurrency contract). apply: When ``False`` (default) return the planned change without a write. When ``True`` issue the PUT and return the applied result. Returns: A :class:`PlannedPageUpdate`. ``applied`` is ``False`` for a dry-run, ``True`` after a successful PUT. Raises: ConfluenceError: On auth failure, on a failed current-body read, or on a non-2xx PUT response (apply path only). """ new_version = version_number + 1 # Compute the body delta against the page's current storage body. The # GET is the only read; it never mutates, so it is safe in dry-run. current_body = "" try: current_page = self.get_page(page_id) current_body = _extract_storage_body(current_page) except ConfluenceError: # If we cannot read the current body we still produce a plan, but the # diff is against an empty baseline (whole new body shown as added). current_body = "" body_delta = body_diff(current_body, body_storage, page_id=page_id) if not apply: return PlannedPageUpdate( page_id=page_id, title=title, current_version=version_number, new_version=new_version, body_storage=body_storage, body_delta=body_delta, applied=False, ) auth = self._ensure_auth() # Defense-in-depth: percent-encode the id segment so a malformed id can # never rewrite the request path (request-path injection). id_segment = _urlparse.quote(str(page_id), safe="") url = f"{auth.base}/wiki/api/v2/pages/{id_segment}" payload = { "id": str(page_id), "status": "current", "title": title, "body": {"representation": "storage", "value": body_storage}, "version": {"number": new_version}, } data = json.dumps(payload).encode("utf-8") status, body = self._http.put( url, headers=auth.headers(extra={"Content-Type": "application/json"}), data=data, ) if not (200 <= status < 300): raise ConfluenceError(status, _stringify(body)) return PlannedPageUpdate( page_id=page_id, title=title, current_version=version_number, new_version=new_version, body_storage=body_storage, body_delta=body_delta, applied=True, ) # --------------------------------------------------------------------------- # Small helpers. # --------------------------------------------------------------------------- def count_storage_macros(body: str) -> int: """Count Confluence macro elements in a storage-format body. Mermaid diagrams and other extensions live in storage XHTML as ```` / ```` elements whose payload is lost if the body is wholesale-replaced. Counting them lets a writer refuse a storage update that would drop macros (the page-1540098 diagram-loss class). Returns ``0`` for an empty/None body. """ if not body: return 0 return len(_STORAGE_MACRO_RE.findall(body)) def storage_macro_signature(body: str) -> Counter[str]: """Return a multiset of macro IDENTITIES in a storage-format body. Each ```` is keyed by its ``ac:name`` (e.g. ``"macro:mermaid-cloud"``), unnamed macros under ``"macro:_unnamed"``, and ```` under ``"adf-extension"``. Comparing the current page's signature to a proposed body's lets a writer refuse a storage update that DROPS a specific diagram macro even when an unrelated macro keeps the raw count equal (an identity-blind count guard would miss that). Empty/None body yields an empty ``Counter``. """ sig: Counter[str] = Counter() if not body: return sig for kind, attrs in _STORAGE_MACRO_TAG_RE.findall(body): if kind.lower() == "adf-extension": sig["adf-extension"] += 1 continue name_match = _AC_NAME_RE.search(attrs) key = ( f"macro:{name_match.group(1).strip().lower()}" if name_match else "macro:_unnamed" ) sig[key] += 1 return sig def _extract_storage_body(page: dict[str, Any]) -> str: """Pull ``body.storage.value`` from a v2 page object, defaulting to ``""``.""" body = page.get("body") if isinstance(body, dict): storage = body.get("storage") if isinstance(storage, dict): value = storage.get("value") if isinstance(value, str): return value return "" def _stringify(body: dict[str, Any] | bytes) -> str: """Render a response body for error messages (JSON or decoded bytes).""" if isinstance(body, dict): return json.dumps(body) if isinstance(body, bytes): return body.decode("utf-8", "replace") return str(body)