"""Bedrock AI-fallback extraction for the work-order email processor. Owns the Bedrock model id, the -tag neutralizer, and the extract_with_bedrock call. The boto3 bedrock-runtime client is built lazily on first use so tests can patch this module's ``bedrock`` attribute before any real client is constructed (moto-before-handler invariant). """ import json import os import re from decimal import Decimal import boto3 from prompts import EXTRACTION_PROMPT BEDROCK_MODEL_ID = os.environ.get( "BEDROCK_MODEL_ID", "us.anthropic.claude-haiku-4-5-20251001-v1:0" ) # An / (or whitespace-padded variant) appearing INSIDE the # untrusted email text could forge the data-block boundary, so any such # sequence is neutralized before wrapping. A single [\s/]* class (not two # \s* around an optional /) keeps matching linear -- the two-quantifier form # backtracks quadratically on "<" + a long whitespace run (attacker DoS). _EMAIL_TAG_RE = re.compile(r"<[\s/]*email\b", re.IGNORECASE) # Lazy cached Bedrock client. Keeps the public attribute name ``bedrock`` so the # test monkeypatch target changes module only, not attribute name. bedrock = None def _get_bedrock(): global bedrock if bedrock is None: bedrock = boto3.client("bedrock-runtime") return bedrock def extract_with_bedrock(email_data: dict) -> dict: """Send parsed email to Claude on Bedrock for structured extraction. The untrusted email body is wrapped in an explicit XML-tagged data block () to delimit data from instructions; -tag lookalikes inside the untrusted text are neutralized so the boundary cannot be forged. The system prompt instructs the model to treat the block as data only, which (combined with the downstream validate_ai_fallback gate) defends against prompt injection from DKIM-passing but attacker-controlled email bodies. """ email_text = ( f"Subject: {email_data['subject']}\n" f"From: {email_data['sender']}\n" f"To: {email_data['to']}\n" f"CC: {email_data['cc']}\n" f"Date: {email_data['date']}\n" f"\n---\n\n" f"{email_data['body']}" ) email_text = _EMAIL_TAG_RE.sub("[email-tag]", email_text) resp = _get_bedrock().invoke_model( modelId=BEDROCK_MODEL_ID, body=json.dumps( { "anthropic_version": "bedrock-2023-05-31", "max_tokens": 1024, # Greedy decoding: retries of the same email should get the # same extraction back (advisory A1). Not a hard guarantee of # determinism, so model output still never enters a table key. "temperature": 0, "messages": [ { "role": "user", "content": ( f"{EXTRACTION_PROMPT}\n\n\n{email_text}\n" ), } ], } ), ) response_text = json.loads(resp["body"].read())["content"][0]["text"] # Extract JSON from response (handle markdown code blocks) json_match = re.search(r"```(?:json)?\s*(.*?)```", response_text, re.DOTALL) if json_match: response_text = json_match.group(1) # parse_float=Decimal is CRITICAL: DynamoDB rejects Python floats. return json.loads(response_text.strip(), parse_float=Decimal)