"""PO Bedrock AI-fallback extraction. Sends the parsed email to Claude on Bedrock for structured extraction when the deterministic template parser misses. The untrusted email body is wrapped in an explicit XML-tagged data block and tag lookalikes are neutralized before the Bedrock call; the downstream validate_ai_fallback gate (run in handler) is the fail-closed check on the raw model output. """ import json import os import re from decimal import Decimal import boto3 from prompts import EXTRACTION_PROMPT BEDROCK_MODEL_ID = os.environ.get( "BEDROCK_MODEL_ID", "us.anthropic.claude-haiku-4-5-20251001-v1:0" ) # Neutralize forged / tags in untrusted bodies before they are # wrapped in the real data block. Single [\s/]* class (NOT two \s* # quantifiers around an optional /) keeps matching linear-time -- two adjacent # unbounded quantifiers invite quadratic backtracking on '<' + a long whitespace # run (ReDoS). Ported from WO #104. _EMAIL_TAG_RE = re.compile(r"<[\s/]*email\b", re.IGNORECASE) # Lazily-built, cached Bedrock client. Kept under the public name ``bedrock`` so # the tests' setattr(extraction, "bedrock", fake) patch surface is unchanged; # building at first CALL (not import) keeps the moto-before-handler invariant # and honors any patched fake (the accessor returns it when non-None). bedrock = None def _get_bedrock(): global bedrock if bedrock is None: bedrock = boto3.client("bedrock-runtime") return bedrock def extract_with_claude(email_data: dict) -> dict: """Send parsed email to Claude on Bedrock for structured extraction. The untrusted email body is wrapped in an explicit XML-tagged data block () to delimit data from instructions; -tag lookalikes inside the untrusted text are neutralized so the boundary cannot be forged. The prompt instructs the model to treat the block as data only, which -- in combination with the downstream validate_ai_fallback gate -- defends against prompt injection from DKIM-passing but attacker-controlled email bodies. """ email_text = ( f"Subject: {email_data['subject']}\n" f"From: {email_data['sender']}\n" f"To: {email_data['to']}\n" f"Date: {email_data['date']}\n" f"\n---\n\n" f"{email_data['body']}" ) # Neutralize forged closing/opening tags BEFORE wrapping, so DKIM-passing but # attacker-controlled content cannot escape the data block. Applied # to the full assembled text -- subject/from/to/date AND body. email_text = _EMAIL_TAG_RE.sub("[email-tag]", email_text) resp = _get_bedrock().invoke_model( modelId=BEDROCK_MODEL_ID, body=json.dumps( { "anthropic_version": "bedrock-2023-05-31", "max_tokens": 2048, # Greedy decoding: retries of the same email should get the # same extraction back. Not a hard determinism guarantee, so # model output still never enters a table key unvalidated (see # validate_ai_fallback). "temperature": 0, "messages": [ { "role": "user", "content": f"{EXTRACTION_PROMPT}\n\n\n{email_text}\n", } ], } ), ) response_text = json.loads(resp["body"].read())["content"][0]["text"] # Extract JSON from response (handle markdown code blocks) json_match = re.search(r"```(?:json)?\s*(.*?)```", response_text, re.DOTALL) if json_match: response_text = json_match.group(1) # parse_float=Decimal is CRITICAL: DynamoDB rejects Python floats. return json.loads(response_text.strip(), parse_float=Decimal)