procurement-ingest/lambdas/wo/email_processor/extraction.py
Adam Moussa d677358801
Some checks are pending
Deploy / deploy (push) Waiting to run
fix(wo): reject non-str AI free-text fields (#158)
2026-08-04 19:39:20 -04:00

90 lines
3.4 KiB
Python

"""Bedrock AI-fallback extraction for the work-order email processor.
Owns the Bedrock model id, the <email>-tag neutralizer, and the
extract_with_bedrock call. The boto3 bedrock-runtime client is built lazily on
first use so tests can patch this module's ``bedrock`` attribute before any real
client is constructed (moto-before-handler invariant).
"""
import json
import os
import re
from decimal import Decimal
import boto3
from prompts import EXTRACTION_PROMPT
BEDROCK_MODEL_ID = os.environ.get(
"BEDROCK_MODEL_ID", "us.anthropic.claude-haiku-4-5-20251001-v1:0"
)
# An <email>/</email> (or whitespace-padded variant) appearing INSIDE the
# untrusted email text could forge the data-block boundary, so any such
# sequence is neutralized before wrapping. A single [\s/]* class (not two
# \s* around an optional /) keeps matching linear -- the two-quantifier form
# backtracks quadratically on "<" + a long whitespace run (attacker DoS).
_EMAIL_TAG_RE = re.compile(r"<[\s/]*email\b", re.IGNORECASE)
# Lazy cached Bedrock client. Keeps the public attribute name ``bedrock`` so the
# test monkeypatch target changes module only, not attribute name.
bedrock = None
def _get_bedrock():
global bedrock
if bedrock is None:
bedrock = boto3.client("bedrock-runtime")
return bedrock
def extract_with_bedrock(email_data: dict) -> dict:
"""Send parsed email to Claude on Bedrock for structured extraction.
The untrusted email body is wrapped in an explicit XML-tagged data block
(<email>) to delimit data from instructions; <email>-tag lookalikes inside
the untrusted text are neutralized so the boundary cannot be forged. The
system prompt instructs the model to treat the block as data only, which
(combined with the downstream validate_ai_fallback gate) defends against
prompt injection from DKIM-passing but attacker-controlled email bodies.
"""
email_text = (
f"Subject: {email_data['subject']}\n"
f"From: {email_data['sender']}\n"
f"To: {email_data['to']}\n"
f"CC: {email_data['cc']}\n"
f"Date: {email_data['date']}\n"
f"\n---\n\n"
f"{email_data['body']}"
)
email_text = _EMAIL_TAG_RE.sub("[email-tag]", email_text)
resp = _get_bedrock().invoke_model(
modelId=BEDROCK_MODEL_ID,
body=json.dumps(
{
"anthropic_version": "bedrock-2023-05-31",
"max_tokens": 1024,
# Greedy decoding: retries of the same email should get the
# same extraction back (advisory A1). Not a hard guarantee of
# determinism, so model output still never enters a table key.
"temperature": 0,
"messages": [
{
"role": "user",
"content": (
f"{EXTRACTION_PROMPT}\n\n<email>\n{email_text}\n</email>"
),
}
],
}
),
)
response_text = json.loads(resp["body"].read())["content"][0]["text"]
# Extract JSON from response (handle markdown code blocks)
json_match = re.search(r"```(?:json)?\s*(.*?)```", response_text, re.DOTALL)
if json_match:
response_text = json_match.group(1)
# parse_float=Decimal is CRITICAL: DynamoDB rejects Python floats.
return json.loads(response_text.strip(), parse_float=Decimal)