open-swe/agent/utils/prompt_data.py

32 lines
1.1 KiB
Python
Raw Normal View History

"""Escaping for untrusted text embedded in XML-wrapped prompt data blocks."""
from __future__ import annotations
import re
# Closing tags of every XML wrapper used for untrusted data blocks across the
# reviewer prompt and webhook-built run prompts. XML tolerates whitespace
# around the tag name (e.g. `</body >`, `</ body\n>`), so a literal
# `.replace()` of the canonical spelling alone is insufficient — we match each
# end tag whitespace-tolerantly and rewrite it to an inert, human-readable
# form. A shared superset is safe: escaping a closing tag that a given block
# doesn't use only neutralizes attacker-controlled text.
DATA_BLOCK_WRAPPER_TAGS = (
"pr_review_threads",
"thread",
"comment",
"body",
"pr_overview",
"title",
"requester_instructions",
)
_CLOSING_TAG_RE = re.compile(
r"</\s*(" + "|".join(DATA_BLOCK_WRAPPER_TAGS) + r")\s*>",
re.IGNORECASE,
)
def escape_for_data_block(text: str) -> str:
"""Neutralize closing tags so an attacker-controlled body can't break out."""
return _CLOSING_TAG_RE.sub(lambda m: f"</{m.group(1).lower()}_>", text)