""" PO email processor Lambda. Triggered by S3 events when SES delivers a Coupa PO email. Parses the raw email, sends it to Claude for structured extraction, then writes the result to the purchase-orders DynamoDB table. """ import email import json import logging import os import re from datetime import datetime, timezone from decimal import Decimal from email import policy import boto3 from ses_auth import authenticate_inbound_email logger = logging.getLogger() logger.setLevel(logging.INFO) s3 = boto3.client("s3") dynamodb = boto3.resource("dynamodb") bedrock = boto3.client("bedrock-runtime") PO_TABLE = os.environ.get("PO_TABLE", "purchase-orders") BEDROCK_MODEL_ID = os.environ.get( "BEDROCK_MODEL_ID", "us.anthropic.claude-haiku-4-5-20251001-v1:0" ) # "Cancelled" is a sticky, authoritative status: once a PO reaches it, a later # new_po/revision may enrich other fields but must never move it back to a # non-cancelled status. CANCELLED_STATUS = "Cancelled" EXTRACTION_PROMPT = """\ You are an email parser for a purchase order ingest pipeline. The emails are Coupa procurement platform notifications containing purchase order data from Amazon. Analyze the following email and extract structured data. Return ONLY valid JSON with these fields: { "email_type": "new_po" | "revision" | "cancellation", "po_number": "string or null", "po_status": "string or null", "source_system": "coupa", "submitted_by": "string or null", "on_behalf_of": "string or null", "order_date": "string or null", "revision_date": "string or null", "last_opened": "string or null", "acknowledged_at": "string or null", "payment_terms": "string or null", "requisition_number": "string or null", "department": "string or null", "view_order_url": "URL string or null", "supplier": { "name": "string or null" }, "site_code": "string or null", "ship_to": { "name": "string or null", "address": "string or null", "street": "string or null", "city": "string or null", "state": "string or null", "zip": "string or null", "location_code": "string or null", "attn": "string or null" }, "total_amount": 0.0, "currency": "USD", "fiscal_year": "string or null", "trade": "string or null", "coupa_category": "string or null", "line_items": [ { "description": "string", "amount": 0.0, "currency": "USD", "need_by": "date string or null", "category": "string or null", "account_code": "string or null", "period": "string or null", "quantity": "string or null", "unit": "string or null", "price": "string or null" } ] } ## email_type detection - "new_po": email announces a new purchase order being issued - "revision": email announces a revised/updated purchase order (look for "revised" in subject or body) - "cancellation": email announces a PO has been cancelled ## PO number Extract from the email subject or body. Format is a prefix + hyphen + digits: - "2D-18206023", "FK-21088051", "B187-17955555" ## site_code extraction The site code is the Amazon facility code — a 3-5 character alphanumeric code identifying the delivery site. Check these locations in order: 1. Ship-to name in parentheses: "Amazon.com Services LLC (KLAL)" → KLAL 2. Ship-to name after dash: "Amazon.com Services LLC - SNY5" → SNY5 3. Ship-to ATTN line with dash or en-dash: "ATTN: Wagon Wheel DS Station –WKY3" → WKY3 4. Ship-to ATTN line directly: "Attn: HJX1" → HJX1 5. Ship-to name IS the code: if the name is just "DBU2" or similar, use it 6. Line item description prefix: "DYO1 - Sea Haven Ind - Plumbing Repairs" → DYO1 7. Line item description in brackets: "[HMK4] Assemble 3 Wire Security Cages" → HMK4 **Not site codes — do not extract these as site_code:** - RME (Amazon Reliability Maintenance Engineering department) - BBM (Coupa description format tag) - JLL (Jones Lang LaSalle — facilities management vendor) - PARAG, ERIK (vendor/person names) - Industry acronyms: HVAC, LED, PVC, ADA, OSHA, EMR, BMS, DDC, MRO, NTE, EST If the only candidate matches this skip list, set site_code to null. ## Ship-to address parsing Parse the full address into separate fields. Be aware of these common issues: - State abbreviation may be missing entirely (e.g., "Tucson, 85704" with no state) - Zip codes may lack leading zeros (e.g., "MA 2149" should be zip "02149", "NJ 7001" should be "07001") - City names may be misspelled (e.g., "Charoltte" for Charlotte) — extract as-is, do not correct - Format varies: "City, ST - ZIP", "City, ST ZIP", "City, ZIP" (no state) If state cannot be determined from the address, set ship_to.state to null. ## fiscal_year The calendar year the work covers. Determine from: 1. The order_date year (primary source) 2. Need-by dates on line items 3. Year in line item descriptions (e.g., "HVB2 - 2025 - Plumbing PM" → "2025") Use the 4-digit year string (e.g., "2025"). ## trade classification Classify the primary trade from line item descriptions. Use the FIRST match in priority order: **Plumbing - PM**: "plumbing pm", "plumbing preventative", "plumbing maintenance", or BBM format: "Plumbing - Backflow", "Plumbing - Water Heater - Install/Repair" **Plumbing - Reactive**: "plumbing" with: "reactive", "emergency", "repair", "clog", "unclog", "leak", "flood", "sewer", "drain", "grease trap", "jetter", "water line", "toilet", "faucet", "urinal", "pipe" **Electrical**: "electrical", "lighting", "ballast", "outlet", "circuit", "panel", "generator", "transformer", "conduit" (but NOT if "dock door" context) **HVAC**: "hvac", "heating", "cooling", "air conditioning", "RTU", "AHU", "VAV", "refrigerant", "thermostat", "ductwork" **Dock Doors**: "dock door", "dock leveler", "dock plate", "dock seal", "dock bumper" **Doors**: "door", "overhead door", "roll-up", "automatic door", "access door" (only if not matched by Dock Doors above) **Signage**: "sign", "banner", "wayfinding", "marquee", "directional" **Carpentry**: "carpentry", "cabinet", "millwork", "trim", "shelving", "framing" **Fencing/Gates**: "fence", "fencing", "gate", "bollard" (not "dock gate") **Conveyance/MHE**: "conveyor", "MHE", "material handling", "sortation" **Painting**: "paint", "painting", "primer", "coating", "touch-up" **Flooring**: "floor", "tile", "carpet", "epoxy", "polishing" **Janitorial**: "janitorial", "cleaning", "custodial", "pressure wash", "power wash" **Fire/Life Safety**: "fire", "sprinkler", "extinguisher", "fire alarm", "suppression" **Landscaping/Yard**: "landscape", "lawn", "tree", "yard", "mowing", "irrigation" **Roofing**: "roof", "roofing", "gutter", "downspout" **Security/Locksmith**: "lock", "key", "access control", "camera", "security", "CCTV" **Snow Removal**: "snow", "ice", "salt", "de-ice", "plow" **PO Uplift**: description is exactly or primarily "PO Uplift" **General Building - Emergency**: "EMER" prefix, or "emergency" in a general building context **General Building - Handyman**: BBM format "General Building - General Building Technician" **General Building - Project**: BBM format "General Building - General Building Project" **General Building**: any remaining facility maintenance work If a PO has multiple line items with different trades, set "trade" to the primary (non-uplift, non-materials) trade. If genuinely mixed, use the trade of the highest-value line item. ## coupa_category The Coupa commodity/category field if present in the email (e.g., "Maintenance - Facilities", "Plumbing Equipment & Materials"). This is Coupa's own classification, not the trade field. ## General rules - Extract all line items with descriptions, amounts, and metadata - "quantity", "unit" (e.g., "EACH", "HR"), and "price" (unit price) should be extracted when present - total_amount should be the numeric total in USD - If a field is not present in the email, set it to null - Do NOT invent or infer data that is not explicitly in the email """ def parse_raw_email(raw_bytes: bytes) -> dict: """Parse a raw MIME email into subject, sender, and body text.""" msg = email.message_from_bytes(raw_bytes, policy=policy.default) subject = msg.get("Subject", "") sender = msg.get("From", "") to = msg.get("To", "") date = msg.get("Date", "") body = "" if msg.is_multipart(): for part in msg.walk(): content_type = part.get_content_type() if content_type == "text/plain": body = part.get_content() break elif content_type == "text/html" and not body: body = part.get_content() else: body = msg.get_content() return { "subject": subject, "sender": sender, "to": to, "date": date, "body": body, } def extract_with_claude(email_data: dict) -> dict: """Send parsed email to Claude on Bedrock for structured extraction.""" email_text = ( f"Subject: {email_data['subject']}\n" f"From: {email_data['sender']}\n" f"To: {email_data['to']}\n" f"Date: {email_data['date']}\n" f"\n---\n\n" f"{email_data['body']}" ) resp = bedrock.invoke_model( modelId=BEDROCK_MODEL_ID, body=json.dumps( { "anthropic_version": "bedrock-2023-05-31", "max_tokens": 2048, "messages": [ { "role": "user", "content": f"{EXTRACTION_PROMPT}\n\nEMAIL:\n{email_text}", } ], } ), ) response_text = json.loads(resp["body"].read())["content"][0]["text"] # Extract JSON from response (handle markdown code blocks) json_match = re.search(r"```(?:json)?\s*(.*?)```", response_text, re.DOTALL) if json_match: response_text = json_match.group(1) # parse_float=Decimal is CRITICAL: DynamoDB rejects Python floats. return json.loads(response_text.strip(), parse_float=Decimal) def pad_zip(zip_code: str | None) -> str | None: if not zip_code: return zip_code clean = zip_code.strip().split("-")[0] if clean.isdigit() and len(clean) < 5: return clean.zfill(5) + zip_code.strip()[len(clean) :] return zip_code def enrich_parsed(parsed: dict, s3_key: str, email_subject: str): """Add metadata and promote nested fields to top level.""" now = datetime.now(timezone.utc).isoformat() parsed["raw_s3_key"] = s3_key parsed["processed_at"] = now parsed["data_source"] = "email" parsed["email_subject"] = email_subject ship_to = parsed.get("ship_to") or {} if ship_to.get("address"): parsed["ship_to_raw"] = ship_to["address"] if ship_to.get("state"): parsed["state"] = ship_to["state"] if ship_to.get("zip"): ship_to["zip"] = pad_zip(ship_to["zip"]) return parsed def _write_fields(po_number: str, fields: dict, *, guard_cancelled: bool): """SET the given non-null fields on a PO record via update_item. Only the fields supplied are written; absent fields are left untouched, so a partial payload can never delete data that an earlier email established. The record is created if it does not exist (DynamoDB update_item upsert). When ``guard_cancelled`` is True the write carries a ConditionExpression that only permits it while the record is not already Cancelled. The condition is evaluated atomically by DynamoDB at write time, so a cancellation that lands first always wins — there is no read-then-write TOCTOU window. A failed guard raises ConditionalCheckFailedException for the caller to handle. """ table = dynamodb.Table(PO_TABLE) set_parts = [] attr_names = {} attr_values = {} for key, value in fields.items(): if value is None or key == "po_number": continue name_ph = f"#{key}" val_ph = f":{key}" attr_names[name_ph] = key attr_values[val_ph] = value set_parts.append(f"{name_ph} = {val_ph}") if not set_parts: return params = { "Key": {"po_number": po_number}, "UpdateExpression": "SET " + ", ".join(set_parts), "ExpressionAttributeNames": attr_names, "ExpressionAttributeValues": attr_values, } if guard_cancelled: params["ExpressionAttributeValues"][":__cancelled_marker"] = CANCELLED_STATUS params["ConditionExpression"] = ( "attribute_not_exists(po_status) OR po_status <> :__cancelled_marker" ) table.update_item(**params) def _merge_update(po_number: str, fields: dict): """Merge (SET-only) the given fields onto a PO record, keeping Cancelled sticky. Only the fields supplied are written; absent fields are left untouched. The record is created if it does not exist (DynamoDB update_item upsert). "Cancelled" is a sticky, authoritative status. When the incoming payload carries a non-cancelled ``po_status``, the write is guarded by a ConditionExpression so the status is only applied while the record is not already Cancelled — enforced atomically at write time, eliminating the read-then-write TOCTOU where a concurrently-landing cancellation could be silently un-cancelled. If the guard fails (the PO is already Cancelled), the same fields are re-written WITHOUT po_status/cancelled_at and unconditionally, so the other fields still merge while the Cancelled status stays intact. A payload with no ``po_status``, or one whose status is already "Cancelled", needs no guard — a plain merge is correct. This is what keeps legitimate status updates (non-cancelled PO) and status-less revisions from ever being dropped: the status is only ever suppressed on a true un-cancel transition. """ incoming_status = fields.get("po_status") if incoming_status is None or incoming_status == CANCELLED_STATUS: _write_fields(po_number, fields, guard_cancelled=False) return try: _write_fields(po_number, fields, guard_cancelled=True) except dynamodb.meta.client.exceptions.ConditionalCheckFailedException: logger.info( f"PO {po_number} is Cancelled; suppressing incoming " f"po_status={incoming_status!r} and merging remaining fields" ) enrich_fields = { k: v for k, v in fields.items() if k not in ("po_status", "cancelled_at") } _write_fields(po_number, enrich_fields, guard_cancelled=False) def save_new_po(parsed: dict): """Create a PO, merging into any pre-existing record. Uses a merge update rather than a conditional put so that an out-of-order cancellation (which leaves a Cancelled skeleton) is filled in with the full PO data instead of the new_po being silently dropped. "Cancelled" is a sticky status enforced atomically inside _merge_update: if the PO was already cancelled, the new_po backfills its remaining fields (supplier, line_items, amounts) but never un-cancels it. """ po_number = parsed["po_number"] fields = {k: v for k, v in parsed.items() if v is not None} _merge_update(po_number, fields) logger.info(f"Created/merged PO {po_number}") def save_revision(parsed: dict): """Merge revised data into an existing PO without deleting omitted fields. A revision email often omits unchanged sections (line_items, supplier). The previous full-overwrite put_item permanently dropped those. This SETs only the fields present in the revision, leaving everything else intact. "Cancelled" is a sticky status: a revision may enrich a cancelled PO's fields but must never move it to a non-cancelled status. That invariant is enforced atomically inside _merge_update and applies ONLY to the un-cancel transition — a revision that carries no status change, or one targeting a non-cancelled PO, updates po_status normally. """ po_number = parsed["po_number"] fields = {k: v for k, v in parsed.items() if v is not None} _merge_update(po_number, fields) logger.info(f"Revised PO {po_number}") def save_cancellation(parsed: dict): """Mark a PO Cancelled, creating a minimal skeleton if it doesn't exist yet. If the cancellation arrives before the new_po, the skeleton it creates is later backfilled by save_new_po (which preserves this Cancelled status), so no PO data is lost on out-of-order delivery. """ table = dynamodb.Table(PO_TABLE) table.update_item( Key={"po_number": parsed["po_number"]}, UpdateExpression="SET po_status = :status, cancelled_at = :cancelled_at, raw_s3_key = :s3_key", ExpressionAttributeValues={ ":status": "Cancelled", ":cancelled_at": parsed.get( "processed_at", datetime.now(timezone.utc).isoformat() ), ":s3_key": parsed.get("raw_s3_key", ""), }, ) logger.info(f"Cancelled PO {parsed['po_number']}") def handler(event, context): """Lambda entry point. Triggered by S3 ObjectCreated events.""" for record in event.get("Records", []): bucket = record["s3"]["bucket"]["name"] key = record["s3"]["object"]["key"] s3_key = f"s3://{bucket}/{key}" logger.info(f"Processing email: {s3_key}") response = s3.get_object(Bucket=bucket, Key=key) raw_email = response["Body"].read() # Fail-closed sender authentication (INFRA-107): only mail with an # SES-stamped dkim=pass verdict for an allowlisted domain may create # or update purchase orders. Rejected mail is logged and skipped # without erroring the invocation (no retries / DLQ spam). if not authenticate_inbound_email(raw_email, s3_key): continue email_data = parse_raw_email(raw_email) logger.info(f"Subject: {email_data['subject']}") parsed = extract_with_claude(email_data) logger.info( f"Parsed: type={parsed.get('email_type')}, po={parsed.get('po_number')}" ) if not parsed.get("po_number"): logger.warning(f"No PO number found in email, skipping: {key}") continue parsed = enrich_parsed(parsed, s3_key, email_data["subject"]) email_type = parsed.get("email_type") if email_type == "cancellation": save_cancellation(parsed) elif email_type == "revision": save_revision(parsed) else: save_new_po(parsed) return {"statusCode": 200, "body": "OK"}