Address the verifier's residual on the macro-loss fix: the guard was raw-count based, so a body that DROPPED the real Mermaid macro while ADDING an unrelated macro (equal count) could slip through. Replace count_storage_macros in the guard with storage_macro_signature (per-ac:name multiset) and refuse if ANY macro identity loses occurrences. +2 tests.
633 lines
25 KiB
Python
633 lines
25 KiB
Python
"""Confluence REST v2 client with a dry-run-first page updater.
|
|
|
|
This module ports the auth seam from the security-review bash reference
|
|
(``checkers/confluence-doc.sh`` ``conf_api_init`` / ``conf_get``, lines
|
|
~210-304) into a typed Python client suited to the durable pipeline:
|
|
|
|
Auth (OAuth wins when present), mirroring the bash ``conf_api_init`` precedence:
|
|
|
|
A. **2LO client-credentials (OAuth).** When ``CONFLUENCE_OAUTH_CLIENT_ID`` and
|
|
``CONFLUENCE_OAUTH_CLIENT_SECRET`` are set, POST a
|
|
``grant_type=client_credentials`` request to ``CONFLUENCE_OAUTH_TOKEN_URL``
|
|
(default ``https://auth.atlassian.com/oauth/token``) for a Bearer token,
|
|
then resolve the cloudId from ``CONFLUENCE_CLOUD_ID`` or the OAuth-native
|
|
``accessible-resources`` endpoint. Base becomes
|
|
``https://api.atlassian.com/ex/confluence/<cloudId>``.
|
|
B. **Basic auth.** When ``CONFLUENCE_BASE_URL`` + ``CONFLUENCE_EMAIL`` +
|
|
``CONFLUENCE_API_TOKEN`` are set, use HTTP Basic against the configured
|
|
base.
|
|
|
|
As in the bash reference, a missing/incomplete cred set or a failed token /
|
|
cloudId resolution surfaces cleanly (:class:`ConfluenceError`) so the caller can
|
|
SKIP rather than raise a false alarm — mirroring ``conf_api_init`` returning
|
|
non-zero.
|
|
|
|
I/O seam: HTTP is the injected :class:`HttpClient` protocol so tests pass an
|
|
in-memory fake and no live network is touched in any code path the tests hit.
|
|
The default :class:`UrllibHttpClient` is stdlib-only (``urllib``), matching
|
|
``agent_team.transport.github_adapter`` — no third-party dependency.
|
|
|
|
Secrets (``CONFLUENCE_*``) are read from ``os.environ`` at CALL time via
|
|
:func:`init_auth`, never captured at import or stored long-lived on the module.
|
|
|
|
Dry-run-first updater: :meth:`ConfluenceClient.update_page` defaults to
|
|
``apply=False``. In dry-run it returns a :class:`PlannedPageUpdate` (target id,
|
|
``new_version = current + 1``, and a unified-diff body delta) and performs NO
|
|
network write. Only ``apply=True`` issues the PUT.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import base64
|
|
import difflib
|
|
import json
|
|
import os
|
|
import re
|
|
from collections import Counter
|
|
from dataclasses import dataclass, field
|
|
from typing import Any, Protocol
|
|
from urllib import error as _urlerror
|
|
from urllib import parse as _urlparse
|
|
from urllib import request as _urlrequest
|
|
|
|
__all__ = [
|
|
"DEFAULT_OAUTH_TOKEN_URL",
|
|
"ATLASSIAN_API_BASE",
|
|
"ACCESSIBLE_RESOURCES_URL",
|
|
"ConfluenceAuth",
|
|
"ConfluenceClient",
|
|
"ConfluenceError",
|
|
"HttpClient",
|
|
"PlannedPageUpdate",
|
|
"UrllibHttpClient",
|
|
"body_diff",
|
|
"count_storage_macros",
|
|
"init_auth",
|
|
"storage_macro_signature",
|
|
]
|
|
|
|
# Confluence storage-format macros (Mermaid diagrams, info panels, etc.) are
|
|
# serialised as ``<ac:structured-macro>`` / ``<ac:adf-extension>`` elements whose
|
|
# payload does NOT survive a wholesale body replacement. Counting them lets a
|
|
# writer refuse a storage update that would silently drop diagram macros — the
|
|
# exact failure that erased every diagram on page 1540098. Case-insensitive.
|
|
_STORAGE_MACRO_RE = re.compile(
|
|
r"<ac:(?:structured-macro|adf-extension)\b", re.IGNORECASE
|
|
)
|
|
# Opening tag of a macro element + its attribute blob, so the macro's IDENTITY
|
|
# (its ``ac:name``) can be read — an identity-aware guard refuses a body that
|
|
# DROPS a specific diagram macro even when an unrelated macro keeps the raw count
|
|
# equal. ``[^>]*`` matches both self-closing and paired opening tags.
|
|
_STORAGE_MACRO_TAG_RE = re.compile(
|
|
r"<ac:(structured-macro|adf-extension)\b([^>]*)>", re.IGNORECASE
|
|
)
|
|
_AC_NAME_RE = re.compile(r'ac:name\s*=\s*"([^"]*)"', re.IGNORECASE)
|
|
|
|
# Default 2LO token endpoint (overridable via CONFLUENCE_OAUTH_TOKEN_URL).
|
|
DEFAULT_OAUTH_TOKEN_URL = "https://auth.atlassian.com/oauth/token"
|
|
# OAuth-native gateway base for per-cloud Confluence access.
|
|
ATLASSIAN_API_BASE = "https://api.atlassian.com/ex/confluence"
|
|
# OAuth-native resource enumeration used to resolve the cloudId.
|
|
ACCESSIBLE_RESOURCES_URL = "https://api.atlassian.com/oauth/token/accessible-resources"
|
|
|
|
|
|
class ConfluenceError(RuntimeError):
|
|
"""Raised when auth resolution or a REST call fails.
|
|
|
|
Carries an HTTP-ish ``status`` (``0`` for pre-flight/config failures such as
|
|
missing creds or an unresolvable cloudId) and a (truncated) ``body``. A
|
|
caller that wants the bash ``conf_api_init``-style "skip, no false alarm"
|
|
behaviour can catch this and skip rather than propagate.
|
|
"""
|
|
|
|
def __init__(self, status: int, body: str) -> None:
|
|
self.status = status
|
|
self.body = body
|
|
super().__init__(f"Confluence error {status}: {body[:200]}")
|
|
|
|
|
|
class HttpClient(Protocol):
|
|
"""Injected HTTP seam returning ``(status, body)`` for each verb.
|
|
|
|
``body`` is the parsed JSON (a ``dict``) when the response carries JSON, else
|
|
the raw ``bytes``. Keeping the surface narrow (``get``/``post``/``put``)
|
|
means the client has no hard HTTP dependency and tests inject a pure
|
|
in-memory fake — no live network on any tested path.
|
|
"""
|
|
|
|
def get(
|
|
self, url: str, *, headers: dict[str, str]
|
|
) -> tuple[int, dict[str, Any] | bytes]: ...
|
|
|
|
def post(
|
|
self,
|
|
url: str,
|
|
*,
|
|
headers: dict[str, str],
|
|
data: bytes,
|
|
) -> tuple[int, dict[str, Any] | bytes]: ...
|
|
|
|
def put(
|
|
self,
|
|
url: str,
|
|
*,
|
|
headers: dict[str, str],
|
|
data: bytes,
|
|
) -> tuple[int, dict[str, Any] | bytes]: ...
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Default stdlib-only HTTP impl (no third-party dependency at import time).
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _parse_body(raw: bytes, content_type: str) -> dict[str, Any] | bytes:
|
|
"""Parse a response body to a ``dict`` when it is JSON, else return bytes."""
|
|
if not raw:
|
|
return {}
|
|
if "application/json" in (content_type or "").lower():
|
|
try:
|
|
return json.loads(raw.decode("utf-8"))
|
|
except (ValueError, UnicodeDecodeError):
|
|
return raw
|
|
return raw
|
|
|
|
|
|
class UrllibHttpClient:
|
|
"""Stdlib-only :class:`HttpClient` (``urllib``), built lazily.
|
|
|
|
Used only when no client is injected and an actual request is made; tests
|
|
never reach this path because they inject a fake. Matches the
|
|
``github_adapter`` pattern: urls are built from fixed ``https`` Atlassian
|
|
bases, so there is no SSRF/``file://`` surface.
|
|
"""
|
|
|
|
def _request(
|
|
self,
|
|
url: str,
|
|
*,
|
|
method: str,
|
|
headers: dict[str, str],
|
|
data: bytes | None,
|
|
) -> tuple[int, dict[str, Any] | bytes]:
|
|
request = _urlrequest.Request(url, data=data, method=method)
|
|
for key, value in headers.items():
|
|
request.add_header(key, value)
|
|
try:
|
|
with _urlrequest.urlopen(request) as response: # noqa: S310 (trusted atlassian host); nosemgrep
|
|
status = response.getcode()
|
|
raw = response.read()
|
|
content_type = response.headers.get("Content-Type", "")
|
|
except _urlerror.HTTPError as exc: # pragma: no cover - network path
|
|
raw = exc.read()
|
|
content_type = exc.headers.get("Content-Type", "") if exc.headers else ""
|
|
return exc.code, _parse_body(raw, content_type)
|
|
return status, _parse_body(raw, content_type)
|
|
|
|
def get(
|
|
self, url: str, *, headers: dict[str, str]
|
|
) -> tuple[int, dict[str, Any] | bytes]:
|
|
return self._request(url, method="GET", headers=headers, data=None)
|
|
|
|
def post(
|
|
self, url: str, *, headers: dict[str, str], data: bytes
|
|
) -> tuple[int, dict[str, Any] | bytes]:
|
|
return self._request(url, method="POST", headers=headers, data=data)
|
|
|
|
def put(
|
|
self, url: str, *, headers: dict[str, str], data: bytes
|
|
) -> tuple[int, dict[str, Any] | bytes]:
|
|
return self._request(url, method="PUT", headers=headers, data=data)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Auth resolution (pure-ish: env in, resolved config out; one HTTP seam for OAuth).
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ConfluenceAuth:
|
|
"""Resolved auth context: the base URL plus per-request headers.
|
|
|
|
Exactly one of the two modes is materialised (``"oauth"`` or ``"basic"``),
|
|
mirroring ``conf_api_init`` resolving ONE mode. ``base`` already includes the
|
|
OAuth ``/ex/confluence/<cloudId>`` segment when relevant; both modes then
|
|
share the ``/wiki/api/v2/...`` path suffix exactly as ``conf_get`` does.
|
|
"""
|
|
|
|
mode: str
|
|
base: str
|
|
_headers: dict[str, str] = field(default_factory=dict)
|
|
|
|
def headers(self, *, extra: dict[str, str] | None = None) -> dict[str, str]:
|
|
"""Per-request headers (auth + accept), merged with ``extra``."""
|
|
merged = dict(self._headers)
|
|
if extra:
|
|
merged.update(extra)
|
|
return merged
|
|
|
|
|
|
def _resolve_cloud_id(
|
|
http: HttpClient,
|
|
*,
|
|
bearer: str,
|
|
configured_cloud_id: str | None,
|
|
site_url: str | None,
|
|
) -> str:
|
|
"""Resolve the cloudId, preferring config then accessible-resources.
|
|
|
|
Mirrors the bash fallback: use ``CONFLUENCE_CLOUD_ID`` when set, else query
|
|
the OAuth-native ``accessible-resources`` endpoint and prefer the resource
|
|
whose ``url`` matches the configured site, else the first one.
|
|
"""
|
|
if configured_cloud_id:
|
|
return configured_cloud_id
|
|
|
|
status, body = http.get(
|
|
ACCESSIBLE_RESOURCES_URL,
|
|
headers={"Authorization": f"Bearer {bearer}", "Accept": "application/json"},
|
|
)
|
|
if not (200 <= status < 300) or not isinstance(body, list):
|
|
# bash treats this as "could not resolve cloudId -> skip" (status 0).
|
|
raise ConfluenceError(
|
|
0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)"
|
|
)
|
|
|
|
chosen: str | None = None
|
|
if site_url:
|
|
for resource in body:
|
|
if isinstance(resource, dict) and resource.get("url") == site_url:
|
|
chosen = resource.get("id")
|
|
break
|
|
if not chosen and not site_url:
|
|
# No site_url to disambiguate: auto-resolve ONLY when there is exactly
|
|
# one accessible resource. Silently picking the first of several could
|
|
# target the WRONG Confluence site (a confused-deputy / wrong-blast-radius
|
|
# hazard) — require an explicit CONFLUENCE_CLOUD_ID instead.
|
|
candidates = [r["id"] for r in body if isinstance(r, dict) and r.get("id")]
|
|
if len(candidates) == 1:
|
|
chosen = candidates[0]
|
|
elif len(candidates) > 1:
|
|
raise ConfluenceError(
|
|
0,
|
|
"OAuth: multiple accessible Confluence sites; set "
|
|
"CONFLUENCE_CLOUD_ID (or CONFLUENCE_BASE_URL) to disambiguate",
|
|
)
|
|
if not chosen:
|
|
raise ConfluenceError(
|
|
0, "OAuth: could not resolve cloudId (set CONFLUENCE_CLOUD_ID)"
|
|
)
|
|
return chosen
|
|
|
|
|
|
def init_auth(http: HttpClient, *, env: dict[str, str] | None = None) -> ConfluenceAuth:
|
|
"""Resolve ONE auth mode from the environment at call time (OAuth wins).
|
|
|
|
Reads ``CONFLUENCE_*`` from ``env`` (default ``os.environ``) only when
|
|
invoked, never at import. Raises :class:`ConfluenceError` (``status=0``) on a
|
|
missing/incomplete cred set or a failed token / cloudId resolution — the
|
|
``conf_api_init`` "return non-zero -> caller skips" contract, so a missing
|
|
cred path is cleanly detectable.
|
|
|
|
Args:
|
|
http: HTTP seam (used only for the OAuth token + cloudId calls).
|
|
env: Environment mapping to read creds from; defaults to ``os.environ``.
|
|
|
|
Returns:
|
|
A :class:`ConfluenceAuth` with the resolved base + auth headers.
|
|
"""
|
|
environ = os.environ if env is None else env
|
|
|
|
client_id = environ.get("CONFLUENCE_OAUTH_CLIENT_ID")
|
|
client_secret = environ.get("CONFLUENCE_OAUTH_CLIENT_SECRET")
|
|
site_url = environ.get("CONFLUENCE_BASE_URL")
|
|
|
|
# Mode A: 2LO client-credentials (OAuth wins when its creds are present).
|
|
if client_id and client_secret:
|
|
token_url = environ.get("CONFLUENCE_OAUTH_TOKEN_URL") or DEFAULT_OAUTH_TOKEN_URL
|
|
form = _urlparse.urlencode(
|
|
{
|
|
"client_id": client_id,
|
|
"client_secret": client_secret,
|
|
"grant_type": "client_credentials",
|
|
}
|
|
).encode("utf-8")
|
|
status, body = http.post(
|
|
token_url,
|
|
headers={"Content-Type": "application/x-www-form-urlencoded"},
|
|
data=form,
|
|
)
|
|
bearer = body.get("access_token") if isinstance(body, dict) else None
|
|
if not (200 <= status < 300) or not bearer:
|
|
# bash: "OAuth: token request failed — skipping API (no false alarm)".
|
|
raise ConfluenceError(0, "OAuth: token request failed")
|
|
|
|
cloud_id = _resolve_cloud_id(
|
|
http,
|
|
bearer=bearer,
|
|
configured_cloud_id=environ.get("CONFLUENCE_CLOUD_ID"),
|
|
site_url=site_url,
|
|
)
|
|
return ConfluenceAuth(
|
|
mode="oauth",
|
|
base=f"{ATLASSIAN_API_BASE}/{cloud_id}",
|
|
_headers={
|
|
"Authorization": f"Bearer {bearer}",
|
|
"Accept": "application/json",
|
|
},
|
|
)
|
|
|
|
# Mode B: Basic auth (email + API token against the configured base).
|
|
email = environ.get("CONFLUENCE_EMAIL")
|
|
api_token = environ.get("CONFLUENCE_API_TOKEN")
|
|
if site_url and email and api_token:
|
|
raw = f"{email}:{api_token}".encode("utf-8")
|
|
encoded = base64.b64encode(raw).decode("ascii")
|
|
return ConfluenceAuth(
|
|
mode="basic",
|
|
base=site_url.rstrip("/"),
|
|
_headers={
|
|
"Authorization": f"Basic {encoded}",
|
|
"Accept": "application/json",
|
|
},
|
|
)
|
|
|
|
# Neither mode fully configured -> caller skips (no false alarm).
|
|
raise ConfluenceError(0, "no Confluence credentials configured (OAuth or Basic)")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Pure planning logic (no I/O) — kept separate so it is unit-trivial.
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def body_diff(old_body: str, new_body: str, *, page_id: str) -> str:
|
|
"""Render a unified-diff of a page's storage body (old -> new).
|
|
|
|
Pure string logic with no I/O, so the dry-run plan is testable without any
|
|
HTTP. Empty string when the bodies are identical.
|
|
"""
|
|
old_lines = (old_body or "").splitlines(keepends=True)
|
|
new_lines = (new_body or "").splitlines(keepends=True)
|
|
diff = difflib.unified_diff(
|
|
old_lines,
|
|
new_lines,
|
|
fromfile=f"page/{page_id}@current",
|
|
tofile=f"page/{page_id}@planned",
|
|
)
|
|
return "".join(diff)
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class PlannedPageUpdate:
|
|
"""A dry-run page update: the change that WOULD be applied, no PUT issued.
|
|
|
|
Returned by :meth:`ConfluenceClient.update_page` when ``apply=False`` (the
|
|
default). ``applied`` is ``False`` here; the same dataclass is returned with
|
|
``applied=True`` after a real PUT so callers get a uniform shape.
|
|
"""
|
|
|
|
page_id: str
|
|
title: str
|
|
current_version: int
|
|
new_version: int
|
|
body_storage: str
|
|
body_delta: str
|
|
applied: bool = False
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The client (thin I/O wrapper over the seam + the pure planning logic).
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class ConfluenceClient:
|
|
"""Read pages and plan/apply page updates over the injected HTTP seam.
|
|
|
|
Auth is resolved lazily on first use (or eagerly if a :class:`ConfluenceAuth`
|
|
is injected), reading ``CONFLUENCE_*`` from the environment at call time. The
|
|
HTTP seam is injected so tests stay hermetic.
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
*,
|
|
http: HttpClient | None = None,
|
|
auth: ConfluenceAuth | None = None,
|
|
env: dict[str, str] | None = None,
|
|
) -> None:
|
|
"""Construct the client.
|
|
|
|
Args:
|
|
http: Injected HTTP seam. Defaults to a stdlib-only
|
|
:class:`UrllibHttpClient` built lazily (tests inject a fake).
|
|
auth: Pre-resolved auth context. When omitted, auth is resolved on
|
|
first use from the environment via :func:`init_auth`.
|
|
env: Environment mapping for cred resolution; defaults to
|
|
``os.environ`` (read at call time, never at import).
|
|
"""
|
|
self._http: HttpClient = http or UrllibHttpClient()
|
|
self._auth = auth
|
|
self._env = env
|
|
|
|
def _ensure_auth(self) -> ConfluenceAuth:
|
|
"""Resolve auth on first use; raises :class:`ConfluenceError` if missing."""
|
|
if self._auth is None:
|
|
self._auth = init_auth(self._http, env=self._env)
|
|
return self._auth
|
|
|
|
@staticmethod
|
|
def _as_dict(body: dict[str, Any] | bytes) -> dict[str, Any]:
|
|
"""Coerce a response body to a dict or raise a clean parse error."""
|
|
if isinstance(body, dict):
|
|
return body
|
|
raise ConfluenceError(0, "expected JSON object response, got non-JSON body")
|
|
|
|
def get_page(self, page_id: str) -> dict[str, Any]:
|
|
"""GET ``/wiki/api/v2/pages/{id}?body-format=storage``.
|
|
|
|
Returns the parsed page object (including ``version.number`` and
|
|
``body.storage.value``). Raises :class:`ConfluenceError` on a non-2xx
|
|
response, carrying the status so the caller can branch (e.g. 404 = gone).
|
|
"""
|
|
auth = self._ensure_auth()
|
|
# Defense-in-depth: percent-encode the id segment so a malformed id can
|
|
# never rewrite the request path (request-path injection).
|
|
id_segment = _urlparse.quote(str(page_id), safe="")
|
|
url = f"{auth.base}/wiki/api/v2/pages/{id_segment}?body-format=storage"
|
|
status, body = self._http.get(url, headers=auth.headers())
|
|
if not (200 <= status < 300):
|
|
raise ConfluenceError(status, _stringify(body))
|
|
return self._as_dict(body)
|
|
|
|
def page_has_macros(self, page_id: str) -> bool:
|
|
"""Whether the page's CURRENT storage body carries Confluence macros.
|
|
|
|
Reads the page (storage format) and counts ``<ac:structured-macro>`` /
|
|
``<ac:adf-extension>`` elements (e.g. Mermaid diagram extensions). The
|
|
writer node uses this to refuse a wholesale storage overwrite that would
|
|
drop diagram macros. Raises :class:`ConfluenceError` if the page cannot
|
|
be read — the caller treats an unreadable page as "assume macros" (fail
|
|
closed) rather than overwriting blindly.
|
|
"""
|
|
page = self.get_page(str(page_id))
|
|
return count_storage_macros(_extract_storage_body(page)) > 0
|
|
|
|
def update_page(
|
|
self,
|
|
page_id: str,
|
|
title: str,
|
|
body_storage: str,
|
|
version_number: int,
|
|
*,
|
|
apply: bool = False,
|
|
) -> PlannedPageUpdate:
|
|
"""Plan (default) or apply a page update via ``/wiki/api/v2/pages/{id}``.
|
|
|
|
CRITICAL — dry-run by default. With ``apply=False`` (the default) NO
|
|
network write happens: this returns a :class:`PlannedPageUpdate`
|
|
describing the change (target id, ``new_version = version_number + 1``,
|
|
and a unified-diff ``body_delta`` against the page's current storage
|
|
body). Only ``apply=True`` issues the PUT.
|
|
|
|
Args:
|
|
page_id: Target page id.
|
|
title: New page title (Confluence requires title on update).
|
|
body_storage: New body in Confluence ``storage`` representation.
|
|
version_number: The page's CURRENT version number; the PUT/plan uses
|
|
``version_number + 1`` as the new version (Confluence's
|
|
optimistic-concurrency contract).
|
|
apply: When ``False`` (default) return the planned change without a
|
|
write. When ``True`` issue the PUT and return the applied result.
|
|
|
|
Returns:
|
|
A :class:`PlannedPageUpdate`. ``applied`` is ``False`` for a dry-run,
|
|
``True`` after a successful PUT.
|
|
|
|
Raises:
|
|
ConfluenceError: On auth failure, on a failed current-body read, or
|
|
on a non-2xx PUT response (apply path only).
|
|
"""
|
|
new_version = version_number + 1
|
|
|
|
# Compute the body delta against the page's current storage body. The
|
|
# GET is the only read; it never mutates, so it is safe in dry-run.
|
|
current_body = ""
|
|
try:
|
|
current_page = self.get_page(page_id)
|
|
current_body = _extract_storage_body(current_page)
|
|
except ConfluenceError:
|
|
# If we cannot read the current body we still produce a plan, but the
|
|
# diff is against an empty baseline (whole new body shown as added).
|
|
current_body = ""
|
|
body_delta = body_diff(current_body, body_storage, page_id=page_id)
|
|
|
|
if not apply:
|
|
return PlannedPageUpdate(
|
|
page_id=page_id,
|
|
title=title,
|
|
current_version=version_number,
|
|
new_version=new_version,
|
|
body_storage=body_storage,
|
|
body_delta=body_delta,
|
|
applied=False,
|
|
)
|
|
|
|
auth = self._ensure_auth()
|
|
# Defense-in-depth: percent-encode the id segment so a malformed id can
|
|
# never rewrite the request path (request-path injection).
|
|
id_segment = _urlparse.quote(str(page_id), safe="")
|
|
url = f"{auth.base}/wiki/api/v2/pages/{id_segment}"
|
|
payload = {
|
|
"id": str(page_id),
|
|
"status": "current",
|
|
"title": title,
|
|
"body": {"representation": "storage", "value": body_storage},
|
|
"version": {"number": new_version},
|
|
}
|
|
data = json.dumps(payload).encode("utf-8")
|
|
status, body = self._http.put(
|
|
url,
|
|
headers=auth.headers(extra={"Content-Type": "application/json"}),
|
|
data=data,
|
|
)
|
|
if not (200 <= status < 300):
|
|
raise ConfluenceError(status, _stringify(body))
|
|
|
|
return PlannedPageUpdate(
|
|
page_id=page_id,
|
|
title=title,
|
|
current_version=version_number,
|
|
new_version=new_version,
|
|
body_storage=body_storage,
|
|
body_delta=body_delta,
|
|
applied=True,
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Small helpers.
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def count_storage_macros(body: str) -> int:
|
|
"""Count Confluence macro elements in a storage-format body.
|
|
|
|
Mermaid diagrams and other extensions live in storage XHTML as
|
|
``<ac:structured-macro>`` / ``<ac:adf-extension>`` elements whose payload is
|
|
lost if the body is wholesale-replaced. Counting them lets a writer refuse a
|
|
storage update that would drop macros (the page-1540098 diagram-loss class).
|
|
Returns ``0`` for an empty/None body.
|
|
"""
|
|
if not body:
|
|
return 0
|
|
return len(_STORAGE_MACRO_RE.findall(body))
|
|
|
|
|
|
def storage_macro_signature(body: str) -> Counter[str]:
|
|
"""Return a multiset of macro IDENTITIES in a storage-format body.
|
|
|
|
Each ``<ac:structured-macro>`` is keyed by its ``ac:name`` (e.g.
|
|
``"macro:mermaid-cloud"``), unnamed macros under ``"macro:_unnamed"``, and
|
|
``<ac:adf-extension>`` under ``"adf-extension"``. Comparing the current
|
|
page's signature to a proposed body's lets a writer refuse a storage update
|
|
that DROPS a specific diagram macro even when an unrelated macro keeps the
|
|
raw count equal (an identity-blind count guard would miss that). Empty/None
|
|
body yields an empty ``Counter``.
|
|
"""
|
|
sig: Counter[str] = Counter()
|
|
if not body:
|
|
return sig
|
|
for kind, attrs in _STORAGE_MACRO_TAG_RE.findall(body):
|
|
if kind.lower() == "adf-extension":
|
|
sig["adf-extension"] += 1
|
|
continue
|
|
name_match = _AC_NAME_RE.search(attrs)
|
|
key = (
|
|
f"macro:{name_match.group(1).strip().lower()}"
|
|
if name_match
|
|
else "macro:_unnamed"
|
|
)
|
|
sig[key] += 1
|
|
return sig
|
|
|
|
|
|
def _extract_storage_body(page: dict[str, Any]) -> str:
|
|
"""Pull ``body.storage.value`` from a v2 page object, defaulting to ``""``."""
|
|
body = page.get("body")
|
|
if isinstance(body, dict):
|
|
storage = body.get("storage")
|
|
if isinstance(storage, dict):
|
|
value = storage.get("value")
|
|
if isinstance(value, str):
|
|
return value
|
|
return ""
|
|
|
|
|
|
def _stringify(body: dict[str, Any] | bytes) -> str:
|
|
"""Render a response body for error messages (JSON or decoded bytes)."""
|
|
if isinstance(body, dict):
|
|
return json.dumps(body)
|
|
if isinstance(body, bytes):
|
|
return body.decode("utf-8", "replace")
|
|
return str(body)
|