"""PO Bedrock AI-fallback extraction.
Sends the parsed email to Claude on Bedrock for structured extraction when the
deterministic template parser misses. The untrusted email body is wrapped in an
explicit XML-tagged data block and tag lookalikes are neutralized before the
Bedrock call; the downstream validate_ai_fallback gate (run in handler) is the
fail-closed check on the raw model output.
"""
import json
import os
import re
from decimal import Decimal
import boto3
from prompts import EXTRACTION_PROMPT
BEDROCK_MODEL_ID = os.environ.get(
"BEDROCK_MODEL_ID", "us.anthropic.claude-haiku-4-5-20251001-v1:0"
)
# Neutralize forged / tags in untrusted bodies before they are
# wrapped in the real data block. Single [\s/]* class (NOT two \s*
# quantifiers around an optional /) keeps matching linear-time -- two adjacent
# unbounded quantifiers invite quadratic backtracking on '<' + a long whitespace
# run (ReDoS). Ported from WO #104.
_EMAIL_TAG_RE = re.compile(r"<[\s/]*email\b", re.IGNORECASE)
# Lazily-built, cached Bedrock client. Kept under the public name ``bedrock`` so
# the tests' setattr(extraction, "bedrock", fake) patch surface is unchanged;
# building at first CALL (not import) keeps the moto-before-handler invariant
# and honors any patched fake (the accessor returns it when non-None).
bedrock = None
def _get_bedrock():
global bedrock
if bedrock is None:
bedrock = boto3.client("bedrock-runtime")
return bedrock
def extract_with_claude(email_data: dict) -> dict:
"""Send parsed email to Claude on Bedrock for structured extraction.
The untrusted email body is wrapped in an explicit XML-tagged data block
() to delimit data from instructions; -tag lookalikes inside
the untrusted text are neutralized so the boundary cannot be forged. The
prompt instructs the model to treat the block as data only, which -- in
combination with the downstream validate_ai_fallback gate -- defends against
prompt injection from DKIM-passing but attacker-controlled email bodies.
"""
email_text = (
f"Subject: {email_data['subject']}\n"
f"From: {email_data['sender']}\n"
f"To: {email_data['to']}\n"
f"Date: {email_data['date']}\n"
f"\n---\n\n"
f"{email_data['body']}"
)
# Neutralize forged closing/opening tags BEFORE wrapping, so DKIM-passing but
# attacker-controlled content cannot escape the data block. Applied
# to the full assembled text -- subject/from/to/date AND body.
email_text = _EMAIL_TAG_RE.sub("[email-tag]", email_text)
resp = _get_bedrock().invoke_model(
modelId=BEDROCK_MODEL_ID,
body=json.dumps(
{
"anthropic_version": "bedrock-2023-05-31",
"max_tokens": 2048,
# Greedy decoding: retries of the same email should get the
# same extraction back. Not a hard determinism guarantee, so
# model output still never enters a table key unvalidated (see
# validate_ai_fallback).
"temperature": 0,
"messages": [
{
"role": "user",
"content": f"{EXTRACTION_PROMPT}\n\n\n{email_text}\n",
}
],
}
),
)
response_text = json.loads(resp["body"].read())["content"][0]["text"]
# Extract JSON from response (handle markdown code blocks)
json_match = re.search(r"```(?:json)?\s*(.*?)```", response_text, re.DOTALL)
if json_match:
response_text = json_match.group(1)
# parse_float=Decimal is CRITICAL: DynamoDB rejects Python floats.
return json.loads(response_text.strip(), parse_float=Decimal)