This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/agent_team/confluence/client.py
Adam Moussa f56ee1c58d harden(agent-team): identity-aware macro-preservation guard
Address the verifier's residual on the macro-loss fix: the guard was raw-count
based, so a body that DROPPED the real Mermaid macro while ADDING an unrelated
macro (equal count) could slip through. Replace count_storage_macros in the
guard with storage_macro_signature (per-ac:name multiset) and refuse if ANY
macro identity loses occurrences. +2 tests.
2026-06-25 11:22:07 -04:00

633 lines
25 KiB
Python

"""Confluence REST v2 client with a dry-run-first page updater.
This module ports the auth seam from the security-review bash reference
(``checkers/confluence-doc.sh`` ``conf_api_init`` / ``conf_get``, lines
~210-304) into a typed Python client suited to the durable pipeline:
Auth (OAuth wins when present), mirroring the bash ``conf_api_init`` precedence:
A. **2LO client-credentials (OAuth).** When ``CONFLUENCE_OAUTH_CLIENT_ID`` and
``CONFLUENCE_OAUTH_CLIENT_SECRET`` are set, POST a
``grant_type=client_credentials`` request to ``CONFLUENCE_OAUTH_TOKEN_URL``
(default ``https://auth.atlassian.com/oauth/token``) for a Bearer token,
then resolve the cloudId from ``CONFLUENCE_CLOUD_ID`` or the OAuth-native
``accessible-resources`` endpoint. Base becomes
``https://api.atlassian.com/ex/confluence/<cloudId>``.
B. **Basic auth.** When ``CONFLUENCE_BASE_URL`` + ``CONFLUENCE_EMAIL`` +
``CONFLUENCE_API_TOKEN`` are set, use HTTP Basic against the configured
base.
As in the bash reference, a missing/incomplete cred set or a failed token /
cloudId resolution surfaces cleanly (:class:`ConfluenceError`) so the caller can
SKIP rather than raise a false alarm — mirroring ``conf_api_init`` returning
non-zero.
I/O seam: HTTP is the injected :class:`HttpClient` protocol so tests pass an
in-memory fake and no live network is touched in any code path the tests hit.
The default :class:`UrllibHttpClient` is stdlib-only (``urllib``), matching
``agent_team.transport.github_adapter`` — no third-party dependency.
Secrets (``CONFLUENCE_*``) are read from ``os.environ`` at CALL time via
:func:`init_auth`, never captured at import or stored long-lived on the module.
Dry-run-first updater: :meth:`ConfluenceClient.update_page` defaults to
``apply=False``. In dry-run it returns a :class:`PlannedPageUpdate` (target id,
``new_version = current + 1``, and a unified-diff body delta) and performs NO
network write. Only ``apply=True`` issues the PUT.
"""
from __future__ import annotations
import base64
import difflib
import json
import os
import re
from collections import Counter
from dataclasses import dataclass, field
from typing import Any, Protocol
from urllib import error as _urlerror
from urllib import parse as _urlparse
from urllib import request as _urlrequest
__all__ = [
"DEFAULT_OAUTH_TOKEN_URL",
"ATLASSIAN_API_BASE",
"ACCESSIBLE_RESOURCES_URL",
"ConfluenceAuth",
"ConfluenceClient",
"ConfluenceError",
"HttpClient",
"PlannedPageUpdate",
"UrllibHttpClient",
"body_diff",
"count_storage_macros",
"init_auth",
"storage_macro_signature",
]
# Confluence storage-format macros (Mermaid diagrams, info panels, etc.) are
# serialised as ``<ac:structured-macro>`` / ``<ac:adf-extension>`` elements whose
# payload does NOT survive a wholesale body replacement. Counting them lets a
# writer refuse a storage update that would silently drop diagram macros — the
# exact failure that erased every diagram on page 1540098. Case-insensitive.
_STORAGE_MACRO_RE = re.compile(
r"<ac:(?:structured-macro|adf-extension)\b", re.IGNORECASE
)
# Opening tag of a macro element + its attribute blob, so the macro's IDENTITY
# (its ``ac:name``) can be read — an identity-aware guard refuses a body that
# DROPS a specific diagram macro even when an unrelated macro keeps the raw count
# equal. ``[^>]*`` matches both self-closing and paired opening tags.
_STORAGE_MACRO_TAG_RE = re.compile(
r"<ac:(structured-macro|adf-extension)\b([^>]*)>", re.IGNORECASE
)
_AC_NAME_RE = re.compile(r'ac:name\s*=\s*"([^"]*)"', re.IGNORECASE)
# Default 2LO token endpoint (overridable via CONFLUENCE_OAUTH_TOKEN_URL).
DEFAULT_OAUTH_TOKEN_URL = "https://auth.atlassian.com/oauth/token"
# OAuth-native gateway base for per-cloud Confluence access.
ATLASSIAN_API_BASE = "https://api.atlassian.com/ex/confluence"
# OAuth-native resource enumeration used to resolve the cloudId.
ACCESSIBLE_RESOURCES_URL = "https://api.atlassian.com/oauth/token/accessible-resources"
class ConfluenceError(RuntimeError):
"""Raised when auth resolution or a REST call fails.
Carries an HTTP-ish ``status`` (``0`` for pre-flight/config failures such as
missing creds or an unresolvable cloudId) and a (truncated) ``body``. A
caller that wants the bash ``conf_api_init``-style "skip, no false alarm"
behaviour can catch this and skip rather than propagate.
"""
def __init__(self, status: int, body: str) -> None:
self.status = status
self.body = body
super().__init__(f"Confluence error {status}: {body[:200]}")
class HttpClient(Protocol):
"""Injected HTTP seam returning ``(status, body)`` for each verb.
``body`` is the parsed JSON (a ``dict``) when the response carries JSON, else
the raw ``bytes``. Keeping the surface narrow (``get``/``post``/``put``)
means the client has no hard HTTP dependency and tests inject a pure
in-memory fake — no live network on any tested path.
"""
def get(
self, url: str, *, headers: dict[str, str]
) -> tuple[int, dict[str, Any] | bytes]: ...
def post(
self,
url: str,
*,
headers: dict[str, str],
data: bytes,
) -> tuple[int, dict[str, Any] | bytes]: ...
def put(
self,
url: str,
*,
headers: dict[str, str],
data: bytes,
) -> tuple[int, dict[str, Any] | bytes]: ...
# ---------------------------------------------------------------------------
# Default stdlib-only HTTP impl (no third-party dependency at import time).
# ---------------------------------------------------------------------------
def _parse_body(raw: bytes, content_type: str) -> dict[str, Any] | bytes:
"""Parse a response body to a ``dict`` when it is JSON, else return bytes."""
if not raw:
return {}
if "application/json" in (content_type or "").lower():
try:
return json.loads(raw.decode("utf-8"))
except (ValueError, UnicodeDecodeError):
return raw
return raw
class UrllibHttpClient:
"""Stdlib-only :class:`HttpClient` (``urllib``), built lazily.
Used only when no client is injected and an actual request is made; tests
never reach this path because they inject a fake. Matches the
``github_adapter`` pattern: urls are built from fixed ``https`` Atlassian
bases, so there is no SSRF/``file://`` surface.
"""
def _request(
self,
url: str,
*,
method: str,
headers: dict[str, str],
data: bytes | None,
) -> tuple[int, dict[str, Any] | bytes]:
request = _urlrequest.Request(url, data=data, method=method)
for key, value in headers.items():
request.add_header(key, value)
try:
with _urlrequest.urlopen(request) as response: # noqa: S310 (trusted atlassian host); nosemgrep
status = response.getcode()
raw = response.read()
content_type = response.headers.get("Content-Type", "")
except _urlerror.HTTPError as exc: # pragma: no cover - network path
raw = exc.read()
content_type = exc.headers.get("Content-Type", "") if exc.headers else ""
return exc.code, _parse_body(raw, content_type)
return status, _parse_body(raw, content_type)
def get(
self, url: str, *, headers: dict[str, str]
) -> tuple[int, dict[str, Any] | bytes]:
return self._request(url, method="GET", headers=headers, data=None)
def post(
self, url: str, *, headers: dict[str, str], data: bytes
) -> tuple[int, dict[str, Any] | bytes]:
return self._request(url, method="POST", headers=headers, data=data)
def put(
self, url: str, *, headers: dict[str, str], data: bytes
) -> tuple[int, dict[str, Any] | bytes]:
return self._request(url, method="PUT", headers=headers, data=data)
# ---------------------------------------------------------------------------
# Auth resolution (pure-ish: env in, resolved config out; one HTTP seam for OAuth).
# ---------------------------------------------------------------------------
@dataclass(frozen=True)
class ConfluenceAuth:
"""Resolved auth context: the base URL plus per-request headers.
Exactly one of the two modes is materialised (``"oauth"`` or ``"basic"``),
mirroring ``conf_api_init`` resolving ONE mode. ``base`` already includes the
OAuth ``/ex/confluence/<cloudId>`` segment when relevant; both modes then
share the ``/wiki/api/v2/...`` path suffix exactly as ``conf_get`` does.
"""
mode: str
base: str
_headers: dict[str, str] = field(default_factory=dict)
def headers(self, *, extra: dict[str, str] | None = None) -> dict[str, str]:
"""Per-request headers (auth + accept), merged with ``extra``."""
merged = dict(self._headers)
if extra:
merged.update(extra)
return merged
def _resolve_cloud_id(
http: HttpClient,
*,
bearer: str,
configured_cloud_id: str | None,
site_url: str | None,
) -> str:
"""Resolve the cloudId, preferring config then accessible-resources.
Mirrors the bash fallback: use ``CONFLUENCE_CLOUD_ID`` when set, else query
the OAuth-native ``accessible-resources`` endpoint and prefer the resource
whose ``url`` matches the configured site, else the first one.
"""
if configured_cloud_id:
return configured_cloud_id
status, body = http.get(
ACCESSIBLE_RESOURCES_URL,
headers={"Authorization": f"Bearer {bearer}", "Accept": "application/json"},
)
if not (200 <= status < 300) or not isinstance(body, list):
# bash treats this as "could not resolve cloudId -> skip" (status 0).
raise ConfluenceError(
0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)"
)
chosen: str | None = None
if site_url:
for resource in body:
if isinstance(resource, dict) and resource.get("url") == site_url:
chosen = resource.get("id")
break
if not chosen and not site_url:
# No site_url to disambiguate: auto-resolve ONLY when there is exactly
# one accessible resource. Silently picking the first of several could
# target the WRONG Confluence site (a confused-deputy / wrong-blast-radius
# hazard) — require an explicit CONFLUENCE_CLOUD_ID instead.
candidates = [r["id"] for r in body if isinstance(r, dict) and r.get("id")]
if len(candidates) == 1:
chosen = candidates[0]
elif len(candidates) > 1:
raise ConfluenceError(
0,
"OAuth: multiple accessible Confluence sites; set "
"CONFLUENCE_CLOUD_ID (or CONFLUENCE_BASE_URL) to disambiguate",
)
if not chosen:
raise ConfluenceError(
0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)"
)
return chosen
def init_auth(http: HttpClient, *, env: dict[str, str] | None = None) -> ConfluenceAuth:
"""Resolve ONE auth mode from the environment at call time (OAuth wins).
Reads ``CONFLUENCE_*`` from ``env`` (default ``os.environ``) only when
invoked, never at import. Raises :class:`ConfluenceError` (``status=0``) on a
missing/incomplete cred set or a failed token / cloudId resolution — the
``conf_api_init`` "return non-zero -> caller skips" contract, so a missing
cred path is cleanly detectable.
Args:
http: HTTP seam (used only for the OAuth token + cloudId calls).
env: Environment mapping to read creds from; defaults to ``os.environ``.
Returns:
A :class:`ConfluenceAuth` with the resolved base + auth headers.
"""
environ = os.environ if env is None else env
client_id = environ.get("CONFLUENCE_OAUTH_CLIENT_ID")
client_secret = environ.get("CONFLUENCE_OAUTH_CLIENT_SECRET")
site_url = environ.get("CONFLUENCE_BASE_URL")
# Mode A: 2LO client-credentials (OAuth wins when its creds are present).
if client_id and client_secret:
token_url = environ.get("CONFLUENCE_OAUTH_TOKEN_URL") or DEFAULT_OAUTH_TOKEN_URL
form = _urlparse.urlencode(
{
"client_id": client_id,
"client_secret": client_secret,
"grant_type": "client_credentials",
}
).encode("utf-8")
status, body = http.post(
token_url,
headers={"Content-Type": "application/x-www-form-urlencoded"},
data=form,
)
bearer = body.get("access_token") if isinstance(body, dict) else None
if not (200 <= status < 300) or not bearer:
# bash: "OAuth: token request failed — skipping API (no false alarm)".
raise ConfluenceError(0, "OAuth: token request failed")
cloud_id = _resolve_cloud_id(
http,
bearer=bearer,
configured_cloud_id=environ.get("CONFLUENCE_CLOUD_ID"),
site_url=site_url,
)
return ConfluenceAuth(
mode="oauth",
base=f"{ATLASSIAN_API_BASE}/{cloud_id}",
_headers={
"Authorization": f"Bearer {bearer}",
"Accept": "application/json",
},
)
# Mode B: Basic auth (email + API token against the configured base).
email = environ.get("CONFLUENCE_EMAIL")
api_token = environ.get("CONFLUENCE_API_TOKEN")
if site_url and email and api_token:
raw = f"{email}:{api_token}".encode("utf-8")
encoded = base64.b64encode(raw).decode("ascii")
return ConfluenceAuth(
mode="basic",
base=site_url.rstrip("/"),
_headers={
"Authorization": f"Basic {encoded}",
"Accept": "application/json",
},
)
# Neither mode fully configured -> caller skips (no false alarm).
raise ConfluenceError(0, "no Confluence credentials configured (OAuth or Basic)")
# ---------------------------------------------------------------------------
# Pure planning logic (no I/O) — kept separate so it is unit-trivial.
# ---------------------------------------------------------------------------
def body_diff(old_body: str, new_body: str, *, page_id: str) -> str:
"""Render a unified-diff of a page's storage body (old -> new).
Pure string logic with no I/O, so the dry-run plan is testable without any
HTTP. Empty string when the bodies are identical.
"""
old_lines = (old_body or "").splitlines(keepends=True)
new_lines = (new_body or "").splitlines(keepends=True)
diff = difflib.unified_diff(
old_lines,
new_lines,
fromfile=f"page/{page_id}@current",
tofile=f"page/{page_id}@planned",
)
return "".join(diff)
@dataclass(frozen=True)
class PlannedPageUpdate:
"""A dry-run page update: the change that WOULD be applied, no PUT issued.
Returned by :meth:`ConfluenceClient.update_page` when ``apply=False`` (the
default). ``applied`` is ``False`` here; the same dataclass is returned with
``applied=True`` after a real PUT so callers get a uniform shape.
"""
page_id: str
title: str
current_version: int
new_version: int
body_storage: str
body_delta: str
applied: bool = False
# ---------------------------------------------------------------------------
# The client (thin I/O wrapper over the seam + the pure planning logic).
# ---------------------------------------------------------------------------
class ConfluenceClient:
"""Read pages and plan/apply page updates over the injected HTTP seam.
Auth is resolved lazily on first use (or eagerly if a :class:`ConfluenceAuth`
is injected), reading ``CONFLUENCE_*`` from the environment at call time. The
HTTP seam is injected so tests stay hermetic.
"""
def __init__(
self,
*,
http: HttpClient | None = None,
auth: ConfluenceAuth | None = None,
env: dict[str, str] | None = None,
) -> None:
"""Construct the client.
Args:
http: Injected HTTP seam. Defaults to a stdlib-only
:class:`UrllibHttpClient` built lazily (tests inject a fake).
auth: Pre-resolved auth context. When omitted, auth is resolved on
first use from the environment via :func:`init_auth`.
env: Environment mapping for cred resolution; defaults to
``os.environ`` (read at call time, never at import).
"""
self._http: HttpClient = http or UrllibHttpClient()
self._auth = auth
self._env = env
def _ensure_auth(self) -> ConfluenceAuth:
"""Resolve auth on first use; raises :class:`ConfluenceError` if missing."""
if self._auth is None:
self._auth = init_auth(self._http, env=self._env)
return self._auth
@staticmethod
def _as_dict(body: dict[str, Any] | bytes) -> dict[str, Any]:
"""Coerce a response body to a dict or raise a clean parse error."""
if isinstance(body, dict):
return body
raise ConfluenceError(0, "expected JSON object response, got non-JSON body")
def get_page(self, page_id: str) -> dict[str, Any]:
"""GET ``/wiki/api/v2/pages/{id}?body-format=storage``.
Returns the parsed page object (including ``version.number`` and
``body.storage.value``). Raises :class:`ConfluenceError` on a non-2xx
response, carrying the status so the caller can branch (e.g. 404 = gone).
"""
auth = self._ensure_auth()
# Defense-in-depth: percent-encode the id segment so a malformed id can
# never rewrite the request path (request-path injection).
id_segment = _urlparse.quote(str(page_id), safe="")
url = f"{auth.base}/wiki/api/v2/pages/{id_segment}?body-format=storage"
status, body = self._http.get(url, headers=auth.headers())
if not (200 <= status < 300):
raise ConfluenceError(status, _stringify(body))
return self._as_dict(body)
def page_has_macros(self, page_id: str) -> bool:
"""Whether the page's CURRENT storage body carries Confluence macros.
Reads the page (storage format) and counts ``<ac:structured-macro>`` /
``<ac:adf-extension>`` elements (e.g. Mermaid diagram extensions). The
writer node uses this to refuse a wholesale storage overwrite that would
drop diagram macros. Raises :class:`ConfluenceError` if the page cannot
be read — the caller treats an unreadable page as "assume macros" (fail
closed) rather than overwriting blindly.
"""
page = self.get_page(str(page_id))
return count_storage_macros(_extract_storage_body(page)) > 0
def update_page(
self,
page_id: str,
title: str,
body_storage: str,
version_number: int,
*,
apply: bool = False,
) -> PlannedPageUpdate:
"""Plan (default) or apply a page update via ``/wiki/api/v2/pages/{id}``.
CRITICAL — dry-run by default. With ``apply=False`` (the default) NO
network write happens: this returns a :class:`PlannedPageUpdate`
describing the change (target id, ``new_version = version_number + 1``,
and a unified-diff ``body_delta`` against the page's current storage
body). Only ``apply=True`` issues the PUT.
Args:
page_id: Target page id.
title: New page title (Confluence requires title on update).
body_storage: New body in Confluence ``storage`` representation.
version_number: The page's CURRENT version number; the PUT/plan uses
``version_number + 1`` as the new version (Confluence's
optimistic-concurrency contract).
apply: When ``False`` (default) return the planned change without a
write. When ``True`` issue the PUT and return the applied result.
Returns:
A :class:`PlannedPageUpdate`. ``applied`` is ``False`` for a dry-run,
``True`` after a successful PUT.
Raises:
ConfluenceError: On auth failure, on a failed current-body read, or
on a non-2xx PUT response (apply path only).
"""
new_version = version_number + 1
# Compute the body delta against the page's current storage body. The
# GET is the only read; it never mutates, so it is safe in dry-run.
current_body = ""
try:
current_page = self.get_page(page_id)
current_body = _extract_storage_body(current_page)
except ConfluenceError:
# If we cannot read the current body we still produce a plan, but the
# diff is against an empty baseline (whole new body shown as added).
current_body = ""
body_delta = body_diff(current_body, body_storage, page_id=page_id)
if not apply:
return PlannedPageUpdate(
page_id=page_id,
title=title,
current_version=version_number,
new_version=new_version,
body_storage=body_storage,
body_delta=body_delta,
applied=False,
)
auth = self._ensure_auth()
# Defense-in-depth: percent-encode the id segment so a malformed id can
# never rewrite the request path (request-path injection).
id_segment = _urlparse.quote(str(page_id), safe="")
url = f"{auth.base}/wiki/api/v2/pages/{id_segment}"
payload = {
"id": str(page_id),
"status": "current",
"title": title,
"body": {"representation": "storage", "value": body_storage},
"version": {"number": new_version},
}
data = json.dumps(payload).encode("utf-8")
status, body = self._http.put(
url,
headers=auth.headers(extra={"Content-Type": "application/json"}),
data=data,
)
if not (200 <= status < 300):
raise ConfluenceError(status, _stringify(body))
return PlannedPageUpdate(
page_id=page_id,
title=title,
current_version=version_number,
new_version=new_version,
body_storage=body_storage,
body_delta=body_delta,
applied=True,
)
# ---------------------------------------------------------------------------
# Small helpers.
# ---------------------------------------------------------------------------
def count_storage_macros(body: str) -> int:
"""Count Confluence macro elements in a storage-format body.
Mermaid diagrams and other extensions live in storage XHTML as
``<ac:structured-macro>`` / ``<ac:adf-extension>`` elements whose payload is
lost if the body is wholesale-replaced. Counting them lets a writer refuse a
storage update that would drop macros (the page-1540098 diagram-loss class).
Returns ``0`` for an empty/None body.
"""
if not body:
return 0
return len(_STORAGE_MACRO_RE.findall(body))
def storage_macro_signature(body: str) -> Counter[str]:
"""Return a multiset of macro IDENTITIES in a storage-format body.
Each ``<ac:structured-macro>`` is keyed by its ``ac:name`` (e.g.
``"macro:mermaid-cloud"``), unnamed macros under ``"macro:_unnamed"``, and
``<ac:adf-extension>`` under ``"adf-extension"``. Comparing the current
page's signature to a proposed body's lets a writer refuse a storage update
that DROPS a specific diagram macro even when an unrelated macro keeps the
raw count equal (an identity-blind count guard would miss that). Empty/None
body yields an empty ``Counter``.
"""
sig: Counter[str] = Counter()
if not body:
return sig
for kind, attrs in _STORAGE_MACRO_TAG_RE.findall(body):
if kind.lower() == "adf-extension":
sig["adf-extension"] += 1
continue
name_match = _AC_NAME_RE.search(attrs)
key = (
f"macro:{name_match.group(1).strip().lower()}"
if name_match
else "macro:_unnamed"
)
sig[key] += 1
return sig
def _extract_storage_body(page: dict[str, Any]) -> str:
"""Pull ``body.storage.value`` from a v2 page object, defaulting to ``""``."""
body = page.get("body")
if isinstance(body, dict):
storage = body.get("storage")
if isinstance(storage, dict):
value = storage.get("value")
if isinstance(value, str):
return value
return ""
def _stringify(body: dict[str, Any] | bytes) -> str:
"""Render a response body for error messages (JSON or decoded bytes)."""
if isinstance(body, dict):
return json.dumps(body)
if isinstance(body, bytes):
return body.decode("utf-8", "replace")
return str(body)