"""Confluence REST v2 client with a dry-run-first page updater. This module ports the auth seam from the security-review bash reference (``checkers/confluence-doc.sh`` ``conf_api_init`` / ``conf_get``, lines ~210-304) into a typed Python client suited to the durable pipeline: Auth (OAuth wins when present), mirroring the bash ``conf_api_init`` precedence: A. **2LO client-credentials (OAuth).** When ``CONFLUENCE_OAUTH_CLIENT_ID`` and ``CONFLUENCE_OAUTH_CLIENT_SECRET`` are set, POST a ``grant_type=client_credentials`` request to ``CONFLUENCE_OAUTH_TOKEN_URL`` (default ``https://auth.atlassian.com/oauth/token``) for a Bearer token, then resolve the cloudId from ``CONFLUENCE_CLOUD_ID`` or the OAuth-native ``accessible-resources`` endpoint. Base becomes ``https://api.atlassian.com/ex/confluence/``. B. **Basic auth.** When ``CONFLUENCE_BASE_URL`` + ``CONFLUENCE_EMAIL`` + ``CONFLUENCE_API_TOKEN`` are set, use HTTP Basic against the configured base. As in the bash reference, a missing/incomplete cred set or a failed token / cloudId resolution surfaces cleanly (:class:`ConfluenceError`) so the caller can SKIP rather than raise a false alarm — mirroring ``conf_api_init`` returning non-zero. I/O seam: HTTP is the injected :class:`HttpClient` protocol so tests pass an in-memory fake and no live network is touched in any code path the tests hit. The default :class:`UrllibHttpClient` is stdlib-only (``urllib``), matching ``agent_team.transport.github_adapter`` — no third-party dependency. Secrets (``CONFLUENCE_*``) are read from ``os.environ`` at CALL time via :func:`init_auth`, never captured at import or stored long-lived on the module. Dry-run-first updater: :meth:`ConfluenceClient.update_page` defaults to ``apply=False``. In dry-run it returns a :class:`PlannedPageUpdate` (target id, ``new_version = current + 1``, and a unified-diff body delta) and performs NO network write. Only ``apply=True`` issues the PUT. """ from __future__ import annotations import base64 import difflib import json import os from dataclasses import dataclass, field from typing import Any, Protocol from urllib import error as _urlerror from urllib import parse as _urlparse from urllib import request as _urlrequest __all__ = [ "DEFAULT_OAUTH_TOKEN_URL", "ATLASSIAN_API_BASE", "ACCESSIBLE_RESOURCES_URL", "ConfluenceAuth", "ConfluenceClient", "ConfluenceError", "HttpClient", "PlannedPageUpdate", "UrllibHttpClient", "body_diff", "init_auth", ] # Default 2LO token endpoint (overridable via CONFLUENCE_OAUTH_TOKEN_URL). DEFAULT_OAUTH_TOKEN_URL = "https://auth.atlassian.com/oauth/token" # OAuth-native gateway base for per-cloud Confluence access. ATLASSIAN_API_BASE = "https://api.atlassian.com/ex/confluence" # OAuth-native resource enumeration used to resolve the cloudId. ACCESSIBLE_RESOURCES_URL = "https://api.atlassian.com/oauth/token/accessible-resources" class ConfluenceError(RuntimeError): """Raised when auth resolution or a REST call fails. Carries an HTTP-ish ``status`` (``0`` for pre-flight/config failures such as missing creds or an unresolvable cloudId) and a (truncated) ``body``. A caller that wants the bash ``conf_api_init``-style "skip, no false alarm" behaviour can catch this and skip rather than propagate. """ def __init__(self, status: int, body: str) -> None: self.status = status self.body = body super().__init__(f"Confluence error {status}: {body[:200]}") class HttpClient(Protocol): """Injected HTTP seam returning ``(status, body)`` for each verb. ``body`` is the parsed JSON (a ``dict``) when the response carries JSON, else the raw ``bytes``. Keeping the surface narrow (``get``/``post``/``put``) means the client has no hard HTTP dependency and tests inject a pure in-memory fake — no live network on any tested path. """ def get( self, url: str, *, headers: dict[str, str] ) -> tuple[int, dict[str, Any] | bytes]: ... def post( self, url: str, *, headers: dict[str, str], data: bytes, ) -> tuple[int, dict[str, Any] | bytes]: ... def put( self, url: str, *, headers: dict[str, str], data: bytes, ) -> tuple[int, dict[str, Any] | bytes]: ... # --------------------------------------------------------------------------- # Default stdlib-only HTTP impl (no third-party dependency at import time). # --------------------------------------------------------------------------- def _parse_body(raw: bytes, content_type: str) -> dict[str, Any] | bytes: """Parse a response body to a ``dict`` when it is JSON, else return bytes.""" if not raw: return {} if "application/json" in (content_type or "").lower(): try: return json.loads(raw.decode("utf-8")) except (ValueError, UnicodeDecodeError): return raw return raw class UrllibHttpClient: """Stdlib-only :class:`HttpClient` (``urllib``), built lazily. Used only when no client is injected and an actual request is made; tests never reach this path because they inject a fake. Matches the ``github_adapter`` pattern: urls are built from fixed ``https`` Atlassian bases, so there is no SSRF/``file://`` surface. """ def _request( self, url: str, *, method: str, headers: dict[str, str], data: bytes | None, ) -> tuple[int, dict[str, Any] | bytes]: request = _urlrequest.Request(url, data=data, method=method) for key, value in headers.items(): request.add_header(key, value) try: with _urlrequest.urlopen(request) as response: # noqa: S310 (trusted atlassian host); nosemgrep status = response.getcode() raw = response.read() content_type = response.headers.get("Content-Type", "") except _urlerror.HTTPError as exc: # pragma: no cover - network path raw = exc.read() content_type = exc.headers.get("Content-Type", "") if exc.headers else "" return exc.code, _parse_body(raw, content_type) return status, _parse_body(raw, content_type) def get( self, url: str, *, headers: dict[str, str] ) -> tuple[int, dict[str, Any] | bytes]: return self._request(url, method="GET", headers=headers, data=None) def post( self, url: str, *, headers: dict[str, str], data: bytes ) -> tuple[int, dict[str, Any] | bytes]: return self._request(url, method="POST", headers=headers, data=data) def put( self, url: str, *, headers: dict[str, str], data: bytes ) -> tuple[int, dict[str, Any] | bytes]: return self._request(url, method="PUT", headers=headers, data=data) # --------------------------------------------------------------------------- # Auth resolution (pure-ish: env in, resolved config out; one HTTP seam for OAuth). # --------------------------------------------------------------------------- @dataclass(frozen=True) class ConfluenceAuth: """Resolved auth context: the base URL plus per-request headers. Exactly one of the two modes is materialised (``"oauth"`` or ``"basic"``), mirroring ``conf_api_init`` resolving ONE mode. ``base`` already includes the OAuth ``/ex/confluence/`` segment when relevant; both modes then share the ``/wiki/api/v2/...`` path suffix exactly as ``conf_get`` does. """ mode: str base: str _headers: dict[str, str] = field(default_factory=dict) def headers(self, *, extra: dict[str, str] | None = None) -> dict[str, str]: """Per-request headers (auth + accept), merged with ``extra``.""" merged = dict(self._headers) if extra: merged.update(extra) return merged def _resolve_cloud_id( http: HttpClient, *, bearer: str, configured_cloud_id: str | None, site_url: str | None, ) -> str: """Resolve the cloudId, preferring config then accessible-resources. Mirrors the bash fallback: use ``CONFLUENCE_CLOUD_ID`` when set, else query the OAuth-native ``accessible-resources`` endpoint and prefer the resource whose ``url`` matches the configured site, else the first one. """ if configured_cloud_id: return configured_cloud_id status, body = http.get( ACCESSIBLE_RESOURCES_URL, headers={"Authorization": f"Bearer {bearer}", "Accept": "application/json"}, ) if not (200 <= status < 300) or not isinstance(body, list): # bash treats this as "could not resolve cloudId -> skip" (status 0). raise ConfluenceError( 0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)" ) chosen: str | None = None if site_url: for resource in body: if isinstance(resource, dict) and resource.get("url") == site_url: chosen = resource.get("id") break if not chosen: for resource in body: if isinstance(resource, dict) and resource.get("id"): chosen = resource["id"] break if not chosen: raise ConfluenceError( 0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)" ) return chosen def init_auth(http: HttpClient, *, env: dict[str, str] | None = None) -> ConfluenceAuth: """Resolve ONE auth mode from the environment at call time (OAuth wins). Reads ``CONFLUENCE_*`` from ``env`` (default ``os.environ``) only when invoked, never at import. Raises :class:`ConfluenceError` (``status=0``) on a missing/incomplete cred set or a failed token / cloudId resolution — the ``conf_api_init`` "return non-zero -> caller skips" contract, so a missing cred path is cleanly detectable. Args: http: HTTP seam (used only for the OAuth token + cloudId calls). env: Environment mapping to read creds from; defaults to ``os.environ``. Returns: A :class:`ConfluenceAuth` with the resolved base + auth headers. """ environ = os.environ if env is None else env client_id = environ.get("CONFLUENCE_OAUTH_CLIENT_ID") client_secret = environ.get("CONFLUENCE_OAUTH_CLIENT_SECRET") site_url = environ.get("CONFLUENCE_BASE_URL") # Mode A: 2LO client-credentials (OAuth wins when its creds are present). if client_id and client_secret: token_url = environ.get("CONFLUENCE_OAUTH_TOKEN_URL") or DEFAULT_OAUTH_TOKEN_URL form = _urlparse.urlencode( { "client_id": client_id, "client_secret": client_secret, "grant_type": "client_credentials", } ).encode("utf-8") status, body = http.post( token_url, headers={"Content-Type": "application/x-www-form-urlencoded"}, data=form, ) bearer = body.get("access_token") if isinstance(body, dict) else None if not (200 <= status < 300) or not bearer: # bash: "OAuth: token request failed — skipping API (no false alarm)". raise ConfluenceError(0, "OAuth: token request failed") cloud_id = _resolve_cloud_id( http, bearer=bearer, configured_cloud_id=environ.get("CONFLUENCE_CLOUD_ID"), site_url=site_url, ) return ConfluenceAuth( mode="oauth", base=f"{ATLASSIAN_API_BASE}/{cloud_id}", _headers={ "Authorization": f"Bearer {bearer}", "Accept": "application/json", }, ) # Mode B: Basic auth (email + API token against the configured base). email = environ.get("CONFLUENCE_EMAIL") api_token = environ.get("CONFLUENCE_API_TOKEN") if site_url and email and api_token: raw = f"{email}:{api_token}".encode("utf-8") encoded = base64.b64encode(raw).decode("ascii") return ConfluenceAuth( mode="basic", base=site_url.rstrip("/"), _headers={ "Authorization": f"Basic {encoded}", "Accept": "application/json", }, ) # Neither mode fully configured -> caller skips (no false alarm). raise ConfluenceError(0, "no Confluence credentials configured (OAuth or Basic)") # --------------------------------------------------------------------------- # Pure planning logic (no I/O) — kept separate so it is unit-trivial. # --------------------------------------------------------------------------- def body_diff(old_body: str, new_body: str, *, page_id: str) -> str: """Render a unified-diff of a page's storage body (old -> new). Pure string logic with no I/O, so the dry-run plan is testable without any HTTP. Empty string when the bodies are identical. """ old_lines = (old_body or "").splitlines(keepends=True) new_lines = (new_body or "").splitlines(keepends=True) diff = difflib.unified_diff( old_lines, new_lines, fromfile=f"page/{page_id}@current", tofile=f"page/{page_id}@planned", ) return "".join(diff) @dataclass(frozen=True) class PlannedPageUpdate: """A dry-run page update: the change that WOULD be applied, no PUT issued. Returned by :meth:`ConfluenceClient.update_page` when ``apply=False`` (the default). ``applied`` is ``False`` here; the same dataclass is returned with ``applied=True`` after a real PUT so callers get a uniform shape. """ page_id: str title: str current_version: int new_version: int body_storage: str body_delta: str applied: bool = False # --------------------------------------------------------------------------- # The client (thin I/O wrapper over the seam + the pure planning logic). # --------------------------------------------------------------------------- class ConfluenceClient: """Read pages and plan/apply page updates over the injected HTTP seam. Auth is resolved lazily on first use (or eagerly if a :class:`ConfluenceAuth` is injected), reading ``CONFLUENCE_*`` from the environment at call time. The HTTP seam is injected so tests stay hermetic. """ def __init__( self, *, http: HttpClient | None = None, auth: ConfluenceAuth | None = None, env: dict[str, str] | None = None, ) -> None: """Construct the client. Args: http: Injected HTTP seam. Defaults to a stdlib-only :class:`UrllibHttpClient` built lazily (tests inject a fake). auth: Pre-resolved auth context. When omitted, auth is resolved on first use from the environment via :func:`init_auth`. env: Environment mapping for cred resolution; defaults to ``os.environ`` (read at call time, never at import). """ self._http: HttpClient = http or UrllibHttpClient() self._auth = auth self._env = env def _ensure_auth(self) -> ConfluenceAuth: """Resolve auth on first use; raises :class:`ConfluenceError` if missing.""" if self._auth is None: self._auth = init_auth(self._http, env=self._env) return self._auth @staticmethod def _as_dict(body: dict[str, Any] | bytes) -> dict[str, Any]: """Coerce a response body to a dict or raise a clean parse error.""" if isinstance(body, dict): return body raise ConfluenceError(0, "expected JSON object response, got non-JSON body") def get_page(self, page_id: str) -> dict[str, Any]: """GET ``/wiki/api/v2/pages/{id}?body-format=storage``. Returns the parsed page object (including ``version.number`` and ``body.storage.value``). Raises :class:`ConfluenceError` on a non-2xx response, carrying the status so the caller can branch (e.g. 404 = gone). """ auth = self._ensure_auth() # Defense-in-depth: percent-encode the id segment so a malformed id can # never rewrite the request path (request-path injection). id_segment = _urlparse.quote(str(page_id), safe="") url = f"{auth.base}/wiki/api/v2/pages/{id_segment}?body-format=storage" status, body = self._http.get(url, headers=auth.headers()) if not (200 <= status < 300): raise ConfluenceError(status, _stringify(body)) return self._as_dict(body) def update_page( self, page_id: str, title: str, body_storage: str, version_number: int, *, apply: bool = False, ) -> PlannedPageUpdate: """Plan (default) or apply a page update via ``/wiki/api/v2/pages/{id}``. CRITICAL — dry-run by default. With ``apply=False`` (the default) NO network write happens: this returns a :class:`PlannedPageUpdate` describing the change (target id, ``new_version = version_number + 1``, and a unified-diff ``body_delta`` against the page's current storage body). Only ``apply=True`` issues the PUT. Args: page_id: Target page id. title: New page title (Confluence requires title on update). body_storage: New body in Confluence ``storage`` representation. version_number: The page's CURRENT version number; the PUT/plan uses ``version_number + 1`` as the new version (Confluence's optimistic-concurrency contract). apply: When ``False`` (default) return the planned change without a write. When ``True`` issue the PUT and return the applied result. Returns: A :class:`PlannedPageUpdate`. ``applied`` is ``False`` for a dry-run, ``True`` after a successful PUT. Raises: ConfluenceError: On auth failure, on a failed current-body read, or on a non-2xx PUT response (apply path only). """ new_version = version_number + 1 # Compute the body delta against the page's current storage body. The # GET is the only read; it never mutates, so it is safe in dry-run. current_body = "" try: current_page = self.get_page(page_id) current_body = _extract_storage_body(current_page) except ConfluenceError: # If we cannot read the current body we still produce a plan, but the # diff is against an empty baseline (whole new body shown as added). current_body = "" body_delta = body_diff(current_body, body_storage, page_id=page_id) if not apply: return PlannedPageUpdate( page_id=page_id, title=title, current_version=version_number, new_version=new_version, body_storage=body_storage, body_delta=body_delta, applied=False, ) auth = self._ensure_auth() # Defense-in-depth: percent-encode the id segment so a malformed id can # never rewrite the request path (request-path injection). id_segment = _urlparse.quote(str(page_id), safe="") url = f"{auth.base}/wiki/api/v2/pages/{id_segment}" payload = { "id": str(page_id), "status": "current", "title": title, "body": {"representation": "storage", "value": body_storage}, "version": {"number": new_version}, } data = json.dumps(payload).encode("utf-8") status, body = self._http.put( url, headers=auth.headers(extra={"Content-Type": "application/json"}), data=data, ) if not (200 <= status < 300): raise ConfluenceError(status, _stringify(body)) return PlannedPageUpdate( page_id=page_id, title=title, current_version=version_number, new_version=new_version, body_storage=body_storage, body_delta=body_delta, applied=True, ) # --------------------------------------------------------------------------- # Small helpers. # --------------------------------------------------------------------------- def _extract_storage_body(page: dict[str, Any]) -> str: """Pull ``body.storage.value`` from a v2 page object, defaulting to ``""``.""" body = page.get("body") if isinstance(body, dict): storage = body.get("storage") if isinstance(storage, dict): value = storage.get("value") if isinstance(value, str): return value return "" def _stringify(body: dict[str, Any] | bytes) -> str: """Render a response body for error messages (JSON or decoded bytes).""" if isinstance(body, dict): return json.dumps(body) if isinstance(body, bytes): return body.decode("utf-8", "replace") return str(body)