"""Deterministic template parser for Coupa purchase order emails. Pure module: no boto3, no network. Runs ahead of the AI extraction path in the purchase-order email processor. Only returns a parsed result when it is proven conformant to one of the known Coupa templates; otherwise it fails closed and signals the caller to fall back to the Bedrock AI extractor. Mirrors the WO parser idioms (lambdas/wo/email_processor/template_parser.py): classify_template -> extract -> validate (FAIL CLOSED) -> try_deterministic_parse returning (parsed|None, method, template_id, reason). A failure is NEVER a parsed result. Full-bucket triage of all 3,448 inbound emails (2026-07-16) fixed the scope: T1 coupa_new_po -- 3,294 / 3,448 (95.5%). Subject "***Copy for Reference*** New Purchase Order has been issued". multipart/alternative; the text/plain part is a stable label-delimited layout (PO ID / Status / Order Date / Revision Date / Payment Term / Req # / Submitted By / On Behalf Of / Supplier block / Shipping block w/ Location Code + Attn / a Lines section whose per-line metadata is U+2022-delimited: Need By / Category / Account / Period [/ optional Part Number]). Maps to email_type "new_po". 99.8% single-line-item, 100% USD. T2 coupa_cancellation -- 100 / 3,448 (2.9%). Subject " Purchase Order # has been cancelled". Minimal body; the only field the handler's save_cancellation() needs is po_number. Maps to email_type "cancellation". DELIBERATELY OUT OF SCOPE -> always AI fallback (never template-parsed): * "New Comment on Purchase Order for Amazon" (19 / 3,448) -- an email_type the handler enum does not model; do not fabricate a PO record deterministically. * revision -- 0 distinct emails in 3,448; no template to build. * any multi-line-item new_po (6 / 3,448) -- Lines-array structure unobserved. * any non-USD new_po (0 observed) -- non-USD path entirely unexercised. * non-Coupa senders (35 / 3,448) -- already rejected at the ses_auth layer. Derived fields (site_code, trade, fiscal_year) are NOT computed here: the parser leaves them None and a shared post-stage (handler enrich_parsed(), the pad_zip precedent) fills them identically on both the template and the LLM path, so the gate judges extraction fidelity only. coupa_category is a verbatim label capture, not a classifier, and IS extracted here. Entry point: try_deterministic_parse(email_data) -> (parsed|None, method, template_id, reason_code). """ import re from dataclasses import dataclass from decimal import Decimal, InvalidOperation # --------------------------------------------------------------------------- # IMPLEMENTATION STATUS # [x] contract + recursive skeleton/normalize # [x] classify_template (both templates) # [x] coupa_cancellation extract + gate # [x] coupa_new_po extract -- labeled fields, duplicate-label anchoring, # U+2022 line split, Decimal amounts (built against the scrubbed real # fixture corpus in tests/fixtures/) # [x] coupa_new_po value-level gate rules V1-V13 (see validate()) # --------------------------------------------------------------------------- # The contract keys (23), EXACTLY -- mirrors the AI EXTRACTION_PROMPT fields. CONTRACT_KEYS = ( "email_type", "po_number", "po_status", "source_system", "submitted_by", "on_behalf_of", "order_date", "revision_date", "last_opened", "acknowledged_at", "payment_terms", "requisition_number", "department", "view_order_url", "supplier", "site_code", "ship_to", "total_amount", "currency", "fiscal_year", "trade", "coupa_category", "line_items", ) SUPPLIER_KEYS = ("name",) SHIP_TO_KEYS = ( "name", "address", "street", "city", "state", "zip", "location_code", "attn", ) LINE_ITEM_KEYS = ( "description", "amount", "currency", "need_by", "category", "account_code", "period", "quantity", "unit", "price", ) # Derived fields the parser must leave None (filled by the shared post-stage). DERIVED_KEYS = ("site_code", "trade", "fiscal_year") SOURCE_SYSTEM = "coupa" # email_type per template. _TEMPLATE_EMAIL_TYPE = { "coupa_new_po": "new_po", "coupa_cancellation": "cancellation", } VALID_EMAIL_TYPES = {"new_po", "revision", "cancellation"} # Only these Status strings were observed on new_po (2,223 + 1,071 of 3,294). # Anything else fails closed to the LLM -- NEVER default-to-new_po. NEW_PO_SAFE_STATUSES = {"Issued - Created", "Issued - Scheduled for email"} # The sticky, authoritative cancellation status. MUST stay in sync with # handler.CANCELLED_STATUS -- the marker the sticky-cancel ConditionExpression # writes and compares against. The AI-fallback gate uses it to forbid a # non-cancellation email_type from carrying "Cancelled" in po_status, so an # AI-path new_po/revision cannot cancel a live PO off dispatch (parity with the # template path, which never emits "Cancelled" on a new_po). _CANCELLED_STATUS = "Cancelled" # Subject classifiers (Python unfolds header continuation lines before we see them). _NEW_PO_SUBJECT = re.compile( r"^\*\*\*Copy for Reference\*\*\* New Purchase Order\s+(?P\S+)\s+has been issued$" ) # Fully anchored, symmetric with _NEW_PO_SUBJECT: the whole subject must be # " Purchase Order # has been cancelled" -- the SiteName prefix # is bounded (no '#', no newline, <=80 chars) so a subject that merely *ends* # with the cancellation phrase (e.g. a forwarded/quoted thread, or arbitrary # prefix text before the tail) is NOT misclassified as a cancellation and # routed to the sticky-Cancelled write. Matched with .match (see # classify_template), never .search. _CANCELLATION_SUBJECT = re.compile( r"^(?P[^#\n]{0,80}?)Purchase Order\s+#(?P[A-Z0-9-]+)\s+has been cancelled\s*$" ) # PO number shape, e.g. 2D-21456967, FK-21920384, B187-17955555. _PO_ID_RE = re.compile(r"^[A-Z0-9]{1,6}-\d+$") # AI-fallback PO id shape. Same prefix+hyphen+digits family as _PO_ID_RE, but # HARDENED for the untrusted AI path exactly as WO hardened _WO_ID_RE: [0-9] # not \d (rejects fullwidth Unicode digits like "2D-18206023" that render # like ASCII but are a distinct DynamoDB partition key) and \A...\Z not ^...$ # (rejects trailing-newline lookalikes "2D-18206023\n"). The handler builds the # purchase-orders partition key from po_number (handler _write_fields Key and # save_cancellation), so an injected "123#x" ('#' not in the class) or bare # "123" (no prefix-hyphen) must fail here. Distinct from _PO_ID_RE, which the # template path additionally byte-equals against the subject id -- do NOT touch # _PO_ID_RE or the template-path validate(). _AI_PO_ID_RE = re.compile(r"\A[A-Z0-9]{1,6}-[0-9]+\Z") # U+2022 bullet delimiting per-line metadata in the Lines section. _BULLET = "•" # --- new_po layout patterns (built against the scrubbed real fixture corpus) --- # Money tokens are read ONLY from three anchored contexts: the Lines-section # ' for ' line, the Total-block standalone amount line, and the # Items-summary ' x ' line. NEVER free money-shaped scanning: # the Items summary carries unit-price tokens distinct from line amounts. _MONEY_RE = re.compile(r"\d{1,3}(?:,\d{3})*\.\d{2}") _CURRENCY_RE = re.compile(r"[A-Z]{3}") # Items-summary quantity line, e.g. '1.0 EACH x 55,206.00'. _SUMMARY_ITEM_RE = re.compile( r"^(?P\d+(?:\.\d+)?) (?P[A-Z]+) x (?P\d{1,3}(?:,\d{3})*\.\d{2})$" ) # Lines-block quantity evidence line, e.g. '1.0 EA' (gate rule V13 cross-check). _LINE_QTY_RE = re.compile(r"^(?P\d+(?:\.\d+)?) (?P[A-Z]+)$") # Lines-block description/amount line. GREEDY desc: '.+' binds the LAST ' for ', # so a description containing the word 'for' can never shift the amount. _LINE_DESC_AMT_RE = re.compile( r"^(?P.+) for (?P\d{1,3}(?:,\d{3})*\.\d{2}) (?P[A-Z]{3})$" ) _VIEW_ORDER_URL_RE = re.compile(r"^https://supplier\.coupahost\.com/orders/\S+$") _ORDER_URL_ID_RE = re.compile(r"^https://supplier\.coupahost\.com/orders/(\d+)\b") _LOCATION_CODE_RE = re.compile(r"^Location Code: (?P\d+)$") _ATTN_RE = re.compile(r"^Attn: (?P.+)$") # Ship-to city line immediately preceding 'United States'. _CITY_STATE_ZIP_RE = re.compile( r"^(?P.+), (?P[A-Z]{2}) (?P\d{5}(?:-\d{4})?)$" ) _US_SENTINEL = "United States" _SUPPLIER_MARKER = ( "SEA HAVEN" # case-sensitive; drift becomes fallback, never wrong data ) # More Detail block: label lines, each expected exactly once (gate rule V9). _MORE_DETAIL_FIELDS = { "Department": "department", "Status": "po_status", "Last Opened": "last_opened", "Order Date": "order_date", "Acknowledged At": "acknowledged_at", "Revision Date": "revision_date", "Payment Term": "payment_terms", "Req #": "requisition_number", } _MORE_DETAIL_LABELS = ("PO ID", *_MORE_DETAIL_FIELDS) # Per-line bullet metadata: closed label set, assigned purely by leading label # (longest label first), NEVER by ordinal position -- real data has an optional # 'Part Number' segment between 'Category' and 'Account', and every run starts # with a 'Supplier ' segment. _BULLET_LABELS = ("Part Number", "Need By", "Category", "Account", "Period", "Supplier") _BULLET_FIELDS = { "Need By": "need_by", "Category": "category", "Account": "account_code", "Period": "period", } # Frozen USPS state/territory codes (50 states + DC + territories). _USPS_STATES = frozenset( """AL AK AZ AR CA CO CT DE FL GA HI ID IL IN IA KS KY LA ME MD MA MI MN MS MO MT NE NV NH NJ NM NY NC ND OH OK OR PA RI SC SD TN TX UT VT VA WA WV WI WY DC PR VI GU AS MP""".split() ) # Sentinel: a label was present but its value did not parse. Must FAIL the gate # (present-but-unparseable), distinct from an absent value (None). _UNPARSEABLE = "__UNPARSEABLE__" # --------------------------------------------------------------------------- # Skeleton helpers (recursive -- unlike WO's flat contract) # --------------------------------------------------------------------------- def _empty_line_item(): return {k: None for k in LINE_ITEM_KEYS} def _empty_candidate(): """Full nested skeleton: every contract key present, None where absent.""" cand = {k: None for k in CONTRACT_KEYS} cand["supplier"] = {k: None for k in SUPPLIER_KEYS} cand["ship_to"] = {k: None for k in SHIP_TO_KEYS} cand["line_items"] = [_empty_line_item()] cand["source_system"] = SOURCE_SYSTEM return cand def _normalize(candidate): """Guarantee exact nested key presence before returning.""" out = _empty_candidate() for k in CONTRACT_KEYS: if k in candidate and k not in ("supplier", "ship_to", "line_items"): out[k] = candidate[k] supplier = candidate.get("supplier") or {} out["supplier"] = {k: supplier.get(k) for k in SUPPLIER_KEYS} ship_to = candidate.get("ship_to") or {} out["ship_to"] = {k: ship_to.get(k) for k in SHIP_TO_KEYS} items = candidate.get("line_items") or [{}] out["line_items"] = [{k: (it or {}).get(k) for k in LINE_ITEM_KEYS} for it in items] return out def _clean(value): """Strip trailing CR and U+00A0 nbsp that every captured Coupa value carries; collapse nothing else. Returns None for empty/placeholder 'None'.""" if value is None: return None v = value.replace("\r", "").replace(" ", " ").strip() if v in ("", "None"): return None return v def _plain_lines(body): return body.replace("\r\n", "\n").replace("\r", "\n").split("\n") def _to_decimal(raw): """Parse a Coupa money token ('18,624.05') to Decimal, stripping thousands separators. Returns _UNPARSEABLE if it does not parse (gate must reject).""" if raw is None: return None token = raw.replace(",", "").strip() try: return Decimal(token) except (InvalidOperation, ValueError): return _UNPARSEABLE # --------------------------------------------------------------------------- # Line navigation helpers (shared by the extractor and the gate; both operate # on email_data["body"] only -- the gate re-derives its own byte evidence and # never trusts extractor-carried state it can re-derive) # --------------------------------------------------------------------------- def _visible(line): """True when the raw line carries visible content. The literal placeholder 'None' IS visible (unlike _clean, which maps it to None), so label values of 'None' are found -- not skipped over into the next label line.""" return bool(line.replace("\r", "").replace("\xa0", "").strip()) def _indices(lines, label): """All indices whose _clean-ed content equals the label exactly.""" return [i for i, ln in enumerate(lines) if _clean(ln) == label] def _find_after(lines, label, start): """First index >= start whose _clean-ed content equals label, or None.""" for i in range(start, len(lines)): if _clean(lines[i]) == label: return i return None def _next_visible(lines, idx, end=None): """Index of the first visible line strictly after idx (before end), or None.""" stop = len(lines) if end is None else min(end, len(lines)) for j in range(idx + 1, stop): if _visible(lines[j]): return j return None def _split_bullet_segments(raw_line): """Split a Lines-section metadata line on the bare U+2022 bullet; strip each segment of spaces and nbsp; drop empty segments.""" segments = [] for seg in raw_line.replace("\r", "").split(_BULLET): seg = seg.replace("\xa0", " ").strip() if seg: segments.append(seg) return segments def _match_bullet_label(segment): """(label, value) by leading-label prefix match (longest label first) against the closed _BULLET_LABELS set, or (None, None) if unrecognized.""" for label in sorted(_BULLET_LABELS, key=len, reverse=True): if segment == label: return label, None if segment.startswith(label + " "): return label, segment[len(label) :].strip() return None, None def _walk_leaves(obj): """Yield every scalar leaf of a nested dict/list candidate.""" if isinstance(obj, dict): for v in obj.values(): yield from _walk_leaves(v) elif isinstance(obj, list): for v in obj: yield from _walk_leaves(v) else: yield obj # --------------------------------------------------------------------------- # Subject helpers # --------------------------------------------------------------------------- def classify_template(email_data): """Return (template_id, reason). template_id in {coupa_new_po, coupa_cancellation, unknown}.""" subject = _clean(email_data.get("subject")) or "" if _NEW_PO_SUBJECT.match(subject): return "coupa_new_po", "ok" if _CANCELLATION_SUBJECT.match(subject): return "coupa_cancellation", "ok" return "unknown", "subject_no_match" def _subject_po_id(email_data): """PO number parsed from the subject, or None.""" subject = _clean(email_data.get("subject")) or "" m = _NEW_PO_SUBJECT.match(subject) if m: return m.group("po") m = _CANCELLATION_SUBJECT.match(subject) if m: return m.group("po") return None # --------------------------------------------------------------------------- # T2: coupa_cancellation -> cancellation (simple + safe: po_number only) # --------------------------------------------------------------------------- def extract_cancellation(email_data): """Cancellation carries no PO detail we trust beyond the id; the handler's save_cancellation() only needs po_number. Everything else stays None.""" candidate = _empty_candidate() candidate["email_type"] = "cancellation" candidate["po_number"] = _subject_po_id(email_data) return _normalize(candidate) # --------------------------------------------------------------------------- # T1: coupa_new_po -> new_po # --------------------------------------------------------------------------- def _assign_bullet_metadata(item, raw_line): """Assign the U+2022 metadata segments to the item BY LEADING LABEL. 'Supplier' is recognized but not stored on the item (gate rule V5 proves it byte-equals supplier.name from the body); 'Part Number' is recognized but discarded (LINE_ITEM_KEYS has no slot -- inventing one would break the key_set_mismatch rule and LLM-path shape parity). Unrecognized or duplicate segments are the gate's job to reject (rule V8).""" for segment in _split_bullet_segments(raw_line): label, value = _match_bullet_label(segment) field = _BULLET_FIELDS.get(label) if field and item[field] is None: item[field] = _clean(value) def _extract_summary_section(candidate, lines): """Summary block (start .. 'More Detail'): submitted_by, on_behalf_of, the FIRST 'Supplier' name, the unique view_order_url, and the Items-summary quantity lines. Returns (more_detail_index_or_None, summary_item_matches).""" more_detail = _find_after(lines, "More Detail", 0) summary_end = more_detail if more_detail is not None else len(lines) for label, field in ( ("Submitted By", "submitted_by"), ("On Behalf Of", "on_behalf_of"), ): idx = _find_after(lines, label, 0) if idx is not None and idx < summary_end: j = _next_visible(lines, idx, summary_end) if j is not None: candidate[field] = _clean(lines[j]) sup1 = _find_after(lines, "Supplier", 0) if sup1 is not None and sup1 < summary_end: j = _next_visible(lines, sup1, summary_end) if j is not None: candidate["supplier"]["name"] = _clean(lines[j]) # view_order_url: the unique orders link in the summary (0 or >1 -> None). url_lines = [ _clean(lines[i]) for i in range(summary_end) if _VIEW_ORDER_URL_RE.match(_clean(lines[i]) or "") ] if len(url_lines) == 1: candidate["view_order_url"] = url_lines[0] # Items-summary quantity lines ('1.0 EACH x 55,206.00'), collected in order. summary_items = [ m for i in range(summary_end) if (m := _SUMMARY_ITEM_RE.fullmatch(_clean(lines[i]) or "")) ] return more_detail, summary_items def _extract_more_detail_block(candidate, lines, more_detail): """More Detail block ('More Detail' .. second 'Supplier'): the labeled header fields. Returns the second 'Supplier' index (or None). The FIRST 'Shipping' lives in this block; its value must be the literal 'None' placeholder (gate rule V4 tripwire) and is never used for ship_to.""" sup2 = None if more_detail is not None: sup2 = _find_after(lines, "Supplier", more_detail + 1) md_end = sup2 if sup2 is not None else len(lines) for label, field in _MORE_DETAIL_FIELDS.items(): idx = _find_after(lines, label, more_detail + 1) if idx is not None and idx < md_end: j = _next_visible(lines, idx, md_end) if j is not None: candidate[field] = _clean(lines[j]) return sup2 def _fill_location_attn(ship_to, lines, start, end): """Location Code + Attn lines after the 'United States' sentinel; the first of each wins (mirrors the extractor's None-guarded assignment).""" for j in range(start, end): cl = _clean(lines[j]) or "" lc = _LOCATION_CODE_RE.fullmatch(cl) if lc and ship_to["location_code"] is None: ship_to["location_code"] = lc.group("lc") attn = _ATTN_RE.fullmatch(cl) if attn and ship_to["attn"] is None: ship_to["attn"] = _clean(attn.group("attn")) def _fill_ship_to_address(ship_to, lines, ship2, st_end): """Populate ship_to from the second 'Shipping' block, sentinel-anchored on the 'United States' line that terminates the address.""" name_idx = _next_visible(lines, ship2, st_end) if name_idx is None: return ship_to["name"] = _clean(lines[name_idx]) us_idx = _find_after(lines, _US_SENTINEL, name_idx + 1) if us_idx is None or us_idx >= st_end: return city_idx = us_idx - 1 m = None if city_idx > name_idx: m = _CITY_STATE_ZIP_RE.fullmatch(_clean(lines[city_idx]) or "") if m: ship_to["city"] = m.group("city") ship_to["state"] = m.group("state") ship_to["zip"] = m.group("zip") street = [ _clean(lines[j]) for j in range(name_idx + 1, city_idx) if _visible(lines[j]) ] if street: ship_to["street"] = "\n".join(street) ship_to["address"] = "\n".join( _clean(lines[j]) for j in range(name_idx, us_idx + 1) if _visible(lines[j]) ) _fill_location_attn(ship_to, lines, us_idx + 1, st_end) def _extract_ship_to(candidate, lines, sup2): """ship_to block (second 'Shipping' .. 'Lines'). Returns the 'Lines' anchor index (or None).""" ship2 = _find_after(lines, "Shipping", sup2 + 1) if sup2 is not None else None lines_anchor = _find_after(lines, "Lines", ship2 + 1) if ship2 is not None else None st_end = lines_anchor if lines_anchor is not None else len(lines) if ship2 is not None: _fill_ship_to_address(candidate["ship_to"], lines, ship2, st_end) return lines_anchor def _parse_line_blocks(lines, start, end): """Split the Lines section into U+00A0-delimited blocks and build one line item per non-empty block. EVERY block is extracted, even when >1, so gate rule 7 fires with honest multiline_unsupported data (never silently keep item 0).""" blocks, block = [], [] for j in range(start, end): if lines[j].replace("\r", "") == "\xa0": blocks.append(block) block = [] else: block.append(lines[j]) blocks.append(block) items = [] for block in blocks: visible = [ln for ln in block if _visible(ln)] if not visible: continue item = _empty_line_item() for raw in visible: dm = _LINE_DESC_AMT_RE.fullmatch(_clean(raw) or "") if dm and item["description"] is None: item["description"] = _clean(dm.group("desc")) item["amount"] = _to_decimal(dm.group("amt")) item["currency"] = dm.group("cur") elif _BULLET in raw: _assign_bullet_metadata(item, raw) # The optional ' EA' evidence line is not stored: quantity/ # unit/price come from the Items summary; gate rule V13 cross-checks # the EA line against it from the body. items.append(item) return items def _extract_line_items(candidate, lines, lines_anchor, summary_items): """Lines section ('Lines' .. second 'Total'). Populates line_items, coupa_category, and (single-line only) quantity/unit/price from the Items summary. Returns the second 'Total' index (or None).""" total2 = ( _find_after(lines, "Total", lines_anchor + 1) if lines_anchor is not None else None ) items = [] if lines_anchor is not None: end = total2 if total2 is not None else len(lines) items = _parse_line_blocks(lines, lines_anchor + 1, end) if items: candidate["line_items"] = items # coupa_category is a VERBATIM copy of item 0's Category bullet value. candidate["coupa_category"] = items[0]["category"] if len(summary_items) == 1 and len(items) == 1: m = summary_items[0] items[0]["quantity"] = _to_decimal(m.group("qty")) items[0]["unit"] = m.group("unit") items[0]["price"] = _to_decimal(m.group("price")) return total2 def _extract_total_block(candidate, lines, total2): """Total block (second 'Total' .. end): total_amount + currency.""" if total2 is None: return amt_idx = _next_visible(lines, total2) if amt_idx is None: return token = _clean(lines[amt_idx]) or "" if _MONEY_RE.fullmatch(token): candidate["total_amount"] = _to_decimal(token) cur_idx = _next_visible(lines, amt_idx) if cur_idx is not None: cur_token = _clean(lines[cur_idx]) or "" if _CURRENCY_RE.fullmatch(cur_token): candidate["currency"] = cur_token def extract_new_po(email_data): """Extract the new_po contract from the text/plain body. Permissive capture, section-windowed: each anchor line is located by exact _clean-ed full-line equality, STRICTLY AFTER the previous anchor. Missing anchors leave fields None -- the gate then fails closed. Representation- agnostic: matches only on _clean-ed lines, never on '\\r'-suffixed literals (body line endings are decode-path dependent). Duplicate labels: 'Supplier', 'Shipping', 'Total' each appear TWICE (summary placeholder + detail block; the first 'Shipping' value is literally 'None'). supplier.name anchors on the FIRST 'Supplier'; ship_to on the SECOND 'Shipping'; total on the SECOND 'Total'. Delegated section by section to the _extract_* helpers, threading the anchor indices each stage discovers into the next; DERIVED_KEYS (site_code, trade, fiscal_year) stay None -- filled later by the shared post-stage identically on both paths. """ candidate = _empty_candidate() candidate["email_type"] = "new_po" candidate["po_number"] = _subject_po_id(email_data) lines = _plain_lines(email_data["body"]) more_detail, summary_items = _extract_summary_section(candidate, lines) sup2 = _extract_more_detail_block(candidate, lines, more_detail) lines_anchor = _extract_ship_to(candidate, lines, sup2) total2 = _extract_line_items(candidate, lines, lines_anchor, summary_items) _extract_total_block(candidate, lines, total2) # DERIVED_KEYS intentionally left None (shared post-stage fills them). return _normalize(candidate) # --------------------------------------------------------------------------- # Validation gate -- FAIL CLOSED # --------------------------------------------------------------------------- def _structural_keys_ok(candidate): if set(candidate.keys()) != set(CONTRACT_KEYS): return False if set((candidate.get("supplier") or {}).keys()) != set(SUPPLIER_KEYS): return False if set((candidate.get("ship_to") or {}).keys()) != set(SHIP_TO_KEYS): return False items = candidate.get("line_items") if not isinstance(items, list) or not items: return False return all(set((it or {}).keys()) == set(LINE_ITEM_KEYS) for it in items) def validate(candidate, template_id, email_data): # noqa: C901, PLR0911, PLR0912 """Return (True, 'ok') only if provably conformant; else (False, reason). Every rule must hold. See failClosedGateRules in the investigation report. A linear fail-closed rule ladder with one return per reason code -- the branch count is the rule count. Splitting it further adds indirection, not clarity, so the complexity/return/branch ceilings are suppressed here (the value-level rules V1-V13 ARE decomposed, in _validate_new_po_values).""" # (1) known template if template_id not in _TEMPLATE_EMAIL_TYPE: return False, "subject_no_match" # (2) exact nested key-set if not _structural_keys_ok(candidate): return False, "key_set_mismatch" # (3) derived fields must NOT be populated by the parser for k in DERIVED_KEYS: if candidate.get(k) is not None: return False, "derived_field_set" # (4) email_type matches the template's expected type expected_type = _TEMPLATE_EMAIL_TYPE[template_id] et = candidate.get("email_type") if et not in VALID_EMAIL_TYPES: return False, "missing_required_field" if et != expected_type: return False, "email_type_mismatch" # (5) po_number: valid shape AND byte-equals the subject id subject_po = _subject_po_id(email_data) po = candidate.get("po_number") if not po or not _PO_ID_RE.match(str(po)): return False, "missing_required_field" if po != subject_po: return False, "po_id_mismatch" if template_id == "coupa_cancellation": # Body corroboration: the subject SiteName prefix is free text, so the # anchored subject alone cannot distinguish a genuine Coupa cancellation # from an arbitrary " Purchase Order # has been cancelled" # subject. The real Coupa body independently restates the id in a # "Purchase Order # ... has been cancelled" notice; require that # (with the SAME po_number) before marking a PO sticky-Cancelled, so a # misrouted/near-miss email fails closed to the LLM instead. (SEC review # of PR #105, F1.) body = email_data.get("body") or "" restated = re.search(r"Purchase Order\s+#" + re.escape(str(po)) + r"\b", body) if not restated or "cancelled" not in body.lower(): return False, "cancellation_body_unconfirmed" return True, "ok" # ---- coupa_new_po ---- # (6) status must be one of the confirmed-safe strings. status = candidate.get("po_status") if status not in NEW_PO_SAFE_STATUSES: return False, "unrecognized_status" # (7) single-line only -- multi-line Lines structure is unobserved. if len(candidate.get("line_items") or []) != 1: return False, "multiline_unsupported" # (8) currency must be exactly USD (non-USD path entirely unexercised). if candidate.get("currency") != "USD": return False, "non_usd" # (V1-V13) value-level rules: every byte proof is RE-DERIVED from # email_data["body"] -- the gate never trusts extractor-carried state it # can re-derive, so an extractor bug cannot vouch for itself. return _validate_new_po_values(candidate, email_data) def _money_border_ok(line_text, serialized): """True when `serialized` occurs in the source line with a character before it that is not a digit or comma (the '18,624.05' -> '624.05' kill switch: a truncated capture re-serializes as '624.05', but every occurrence of that string in its source line is preceded by a comma or digit).""" idx = line_text.find(serialized) while idx != -1: prev = line_text[idx - 1] if idx > 0 else "" if prev not in "0123456789,": return True idx = line_text.find(serialized, idx + 1) return False @dataclass class _NewPoAnchorFrame: """Anchor frame for the coupa_new_po value gate. V4 (_build_anchor_frame) proves the duplicate-label anchor layout ONCE and carries the shared byte evidence every later rule re-derives from the body: the section indices, plus the Items-summary matches and the captured line price -- so V13 (_check_qty_unit_price) can re-consume exactly what V1 (_check_money_fidelity) proved, without either rule trusting extractor- carried state it cannot re-derive from email_data["body"].""" lines: list more_detail: int supplier_idxs: list shipping_idxs: list total_idxs: list lines_anchor: int lines_end: int summary_matches: list price: object def _anchor_order_ok( more_detail, supplier_idxs, shipping_idxs, total_idxs, lines_anchor ): """Section ordering: summary Supplier < More Detail < detail Supplier < second Shipping < Lines < second Total; summary Total before More Detail.""" return ( supplier_idxs[0] < more_detail < supplier_idxs[1] < shipping_idxs[1] < lines_anchor < total_idxs[1] ) and (total_idxs[0] < more_detail < shipping_idxs[0] < supplier_idxs[1]) def _first_shipping_value(lines, first_shipping_idx): """Value line under the FIRST 'Shipping' (the dup-label-swap tripwire: a real address here means the layout drifted and ship_to was read from the wrong block; the genuine layout carries the literal 'None').""" if first_shipping_idx + 1 >= len(lines): return "" return lines[first_shipping_idx + 1].replace("\r", "").replace("\xa0", "").strip() def _build_anchor_frame(lines, items): """V4 anchor integrity. Returns (frame, None) when the duplicate-label layout is proven, else (None, 'anchor_violation'). Also precomputes the Items-summary matches and the captured line price onto the frame.""" more_detail_idxs = _indices(lines, "More Detail") supplier_idxs = _indices(lines, "Supplier") shipping_idxs = _indices(lines, "Shipping") total_idxs = _indices(lines, "Total") # The Coupa layout repeats each of Supplier/Shipping/Total exactly twice # (summary placeholder + detail block) -- 2 is a structural constant of the # template, not a tunable magic number. if ( len(more_detail_idxs) != 1 or len(supplier_idxs) != 2 # noqa: PLR2004 or len(shipping_idxs) != 2 # noqa: PLR2004 or len(total_idxs) != 2 # noqa: PLR2004 ): return None, "anchor_violation" more_detail = more_detail_idxs[0] lines_anchor = _find_after(lines, "Lines", shipping_idxs[1] + 1) # lines_anchor is proven non-None before _anchor_order_ok consumes it (the # `or` short-circuits); every index below is safe once the counts hold. if ( lines_anchor is None or _first_shipping_value(lines, shipping_idxs[0]) != "None" or not _anchor_order_ok( more_detail, supplier_idxs, shipping_idxs, total_idxs, lines_anchor ) ): return None, "anchor_violation" summary_matches = [ m for i in range(more_detail) if (m := _SUMMARY_ITEM_RE.fullmatch(_clean(lines[i]) or "")) ] frame = _NewPoAnchorFrame( lines=lines, more_detail=more_detail, supplier_idxs=supplier_idxs, shipping_idxs=shipping_idxs, total_idxs=total_idxs, lines_anchor=lines_anchor, lines_end=total_idxs[1], summary_matches=summary_matches, price=items[0].get("price") if items else None, ) return frame, None def _check_line_item_money(candidate, frame): """V1 line-level money fidelity -> amount_mismatch / non_usd. Every captured line amount must re-locate its raw source token in the body: the token fullmatches the grouped money shape, format(value, ',.2f') byte- equals it, and the character before it is not a digit/comma. Line-level currency is pinned too -- a single non-USD line item fails closed even when the Total block reads USD (the non-USD path is entirely unexercised).""" lines = frame.lines items = candidate.get("line_items") or [] for_matches = [] for j in range(frame.lines_anchor + 1, frame.lines_end): m = _LINE_DESC_AMT_RE.fullmatch(_clean(lines[j]) or "") if m: for_matches.append(m) if len(for_matches) != len(items): return "amount_mismatch" for item, m in zip(items, for_matches): amount = item.get("amount") if not isinstance(amount, Decimal): return "amount_mismatch" serialized = format(amount, ",.2f") if serialized != m.group("amt") or not _money_border_ok(m.string, serialized): return "amount_mismatch" currency = candidate.get("currency") if item.get("currency") != currency or m.group("cur") != currency: return "non_usd" return None def _check_summary_price(frame): """V1 summary-price fidelity -> amount_mismatch. The captured Items-summary price must re-locate its raw token exactly as the line amounts do.""" price = frame.price if price is None: return None if not isinstance(price, Decimal) or len(frame.summary_matches) != 1: return "amount_mismatch" serialized = format(price, ",.2f") if serialized != frame.summary_matches[0].group("price"): return "amount_mismatch" if not _money_border_ok(frame.summary_matches[0].string, serialized): return "amount_mismatch" return None def _check_money_fidelity(candidate, frame): """V1 money fidelity: line-item amounts then the Items-summary price.""" return _check_line_item_money(candidate, frame) or _check_summary_price(frame) def _read_total_tokens(lines, total_idxs): """Read the (amount, currency) token pair under each 'Total' anchor. Returns (tokens, True) or (None, False) on any missing/malformed token.""" total_tokens = [] for t_idx in total_idxs: a_idx = _next_visible(lines, t_idx) if a_idx is None: return None, False token = _clean(lines[a_idx]) or "" if not _MONEY_RE.fullmatch(token): return None, False c_idx = _next_visible(lines, a_idx) cur_token = (_clean(lines[c_idx]) or "") if c_idx is not None else "" if not _CURRENCY_RE.fullmatch(cur_token): return None, False total_tokens.append((token, cur_token)) return total_tokens, True def _check_total_proof(candidate, frame): """V3 dual-Total proof + V2 sum proof -> amount_mismatch. Both Total blocks must carry byte-identical tokens that byte-equal the captured total, and the line amounts must sum to it exactly (no qty*price rule -- partial qtys).""" items = candidate.get("line_items") or [] total = candidate.get("total_amount") if not isinstance(total, Decimal): return "amount_mismatch" total_tokens, ok = _read_total_tokens(frame.lines, frame.total_idxs) if ( not ok or total_tokens[0] != total_tokens[1] or format(total, ",.2f") != total_tokens[1][0] or candidate.get("currency") != total_tokens[1][1] or sum(item["amount"] for item in items) != total ): return "amount_mismatch" return None def _check_supplier_proof(candidate, frame): """V5 supplier proof -> anchor_violation. The supplier name must carry the SEA HAVEN marker and byte-equal its restatements in both the summary and the detail block, and must not collide with the ship_to name.""" lines = frame.lines ship_to = candidate.get("ship_to") or {} supplier_name = (candidate.get("supplier") or {}).get("name") if not supplier_name or _SUPPLIER_MARKER not in supplier_name: return "anchor_violation" det_idx = _next_visible(lines, frame.supplier_idxs[1], frame.shipping_idxs[1]) if det_idx is None or _clean(lines[det_idx]) != supplier_name: return "anchor_violation" sum_idx = _next_visible(lines, frame.supplier_idxs[0], frame.more_detail) if sum_idx is None or _clean(lines[sum_idx]) != supplier_name: return "anchor_violation" if ship_to.get("name") == supplier_name: return "anchor_violation" return None def _check_bullet_line(raw, supplier_name): """One Lines-section bullet metadata line: closed label set, no dupes, the required labels present, and the per-item Supplier segment (V5) byte-equal to supplier_name.""" seen = {} for segment in _split_bullet_segments(raw): label, value = _match_bullet_label(segment) if label is None or label in seen: return "bullet_label_unrecognized" seen[label] = value if set(seen) - { "Supplier", "Need By", "Category", "Account", "Period", "Part Number", }: return "bullet_label_unrecognized" if not {"Supplier", "Need By", "Category", "Account", "Period"} <= set(seen): return "bullet_label_unrecognized" if _clean(seen["Supplier"]) != supplier_name: return "anchor_violation" return None def _check_bullet_discipline(candidate, frame): """V8 bullet discipline -> bullet_label_unrecognized (V5's per-item Supplier-segment proof rides the same walk).""" lines = frame.lines items = candidate.get("line_items") or [] supplier_name = (candidate.get("supplier") or {}).get("name") bullet_lines = [ lines[j] for j in range(frame.lines_anchor + 1, frame.lines_end) if _BULLET in lines[j] ] if len(bullet_lines) != len(items): return "bullet_label_unrecognized" for raw in bullet_lines: reason = _check_bullet_line(raw, supplier_name) if reason: return reason return None def _check_ship_to_required(candidate, frame): """V6 ship_to required fields -> missing_required_field. Required fields present, numeric location_code that re-derives from the body, and attn (if present) matching a body Attn line.""" lines = frame.lines ship_to = candidate.get("ship_to") or {} for field in ("name", "street", "city", "state", "zip", "location_code"): if not ship_to.get(field): return "missing_required_field" if not re.fullmatch(r"\d+", ship_to["location_code"]): return "missing_required_field" lc_values = [ m.group("lc") for j in range(frame.shipping_idxs[1] + 1, frame.lines_anchor) if (m := _LOCATION_CODE_RE.fullmatch(_clean(lines[j]) or "")) ] if ship_to["location_code"] not in lc_values: return "missing_required_field" attn_values = [ _clean(m.group("attn")) for j in range(frame.shipping_idxs[1] + 1, frame.lines_anchor) if (m := _ATTN_RE.fullmatch(_clean(lines[j]) or "")) ] if attn_values: if ship_to.get("attn") not in attn_values: return "missing_required_field" elif ship_to.get("attn") is not None: return "missing_required_field" return None def _check_address_shape(candidate, frame): """V7 address shape -> address_shape_invalid. Gated on the RAW pre- enrichment zip: validate() runs BEFORE enrich_parsed, so a short zip fails closed to the LLM path where pad_zip repairs it (both paths then get identical pad_zip treatment downstream).""" lines = frame.lines ship_to = candidate.get("ship_to") or {} us_idx = _find_after(lines, _US_SENTINEL, frame.shipping_idxs[1] + 1) if us_idx is None or us_idx >= frame.lines_anchor: return "address_shape_invalid" m = _CITY_STATE_ZIP_RE.fullmatch(_clean(lines[us_idx - 1]) or "") if not m: return "address_shape_invalid" if ( m.group("city") != ship_to["city"] or m.group("state") != ship_to["state"] or m.group("zip") != ship_to["zip"] ): return "address_shape_invalid" if ship_to["state"] not in _USPS_STATES: return "address_shape_invalid" if not re.fullmatch(r"\d{5}(-\d{4})?", ship_to["zip"]): return "address_shape_invalid" return None def _check_required_fields(candidate, frame): """V9 required labeled fields -> missing_required_field. Nullable by design: on_behalf_of, department, last_opened, acknowledged_at, revision_date, attn, quantity, unit, price.""" lines = frame.lines items = candidate.get("line_items") or [] for field in ( "po_status", "order_date", "payment_terms", "requisition_number", "submitted_by", "view_order_url", ): if candidate.get(field) is None: return "missing_required_field" for item in items: for field in ( "description", "amount", "currency", "need_by", "category", "account_code", "period", ): if item.get(field) is None: return "missing_required_field" for label in _MORE_DETAIL_LABELS: if len(_indices(lines, label)) != 1: return "missing_required_field" if candidate.get("coupa_category") != items[0].get("category"): return "missing_required_field" return None def _check_po_identity(candidate, frame): """V10 PO identity proofs -> po_id_mismatch. The PO id must restate under 'PO ID', in an 'Amazon Purchase Order #' line, and as the numeric tail of the view_order_url.""" lines = frame.lines po = candidate["po_number"] po_id_idx = _indices(lines, "PO ID")[0] v_idx = _next_visible(lines, po_id_idx) if v_idx is None or _clean(lines[v_idx]) != po: return "po_id_mismatch" if not _indices(lines, f"Amazon Purchase Order #{po}"): return "po_id_mismatch" um = _ORDER_URL_ID_RE.match(candidate.get("view_order_url") or "") if not um or um.group(1) != po.split("-", 1)[1]: return "po_id_mismatch" return None def _check_sentinels(candidate, frame): """V11 sentinel discipline -> unparseable_value. A present-but-unparseable value (the _UNPARSEABLE sentinel) must fail closed.""" for leaf in _walk_leaves(candidate): if isinstance(leaf, str) and leaf == _UNPARSEABLE: return "unparseable_value" return None def _check_hygiene(candidate, frame): """V12 hygiene -> residual_artifact. No captured value may retain a raw CR or nbsp artifact.""" for leaf in _walk_leaves(candidate): if isinstance(leaf, str) and ("\r" in leaf or "\xa0" in leaf): return "residual_artifact" return None def _check_qty_unit_shape(quantity, unit, price): """V13 scalar shape: a positive Decimal quantity, an all-caps unit, and a Decimal price.""" if ( not isinstance(quantity, Decimal) or quantity <= 0 or not unit or not re.fullmatch(r"[A-Z]+", unit) or not isinstance(price, Decimal) ): return "amount_mismatch" return None def _check_ea_line_evidence(lines, frame, quantity): """The Lines-block ' EA' evidence line (when present) must numeric- equal the summary quantity.""" ea_matches = [ m for j in range(frame.lines_anchor + 1, frame.lines_end) if (m := _LINE_QTY_RE.fullmatch(_clean(lines[j]) or "")) ] if not ea_matches: return None if len(ea_matches) != 1: return "amount_mismatch" if _to_decimal(ea_matches[0].group("qty")) != quantity: return "amount_mismatch" return None def _check_qty_unit_price(candidate, frame): """V13 quantity/unit/price coherence -> amount_mismatch. Skipped entirely when all three are absent; otherwise every piece must cohere with the Items-summary line and the Lines-block EA evidence.""" items = candidate.get("line_items") or [] quantity = items[0].get("quantity") unit = items[0].get("unit") price = frame.price if quantity is None and unit is None and price is None: return None reason = _check_qty_unit_shape(quantity, unit, price) if reason: return reason if ( len(frame.summary_matches) != 1 or _to_decimal(frame.summary_matches[0].group("qty")) != quantity or frame.summary_matches[0].group("unit") != unit ): return "amount_mismatch" return _check_ea_line_evidence(frame.lines, frame, quantity) # Value-level rules in spec order, each returning a reason code or None. V4 # builds the shared anchor frame first (below); these consume it. _NEW_PO_VALUE_CHECKS = ( _check_money_fidelity, # V1 _check_total_proof, # V3 + V2 _check_supplier_proof, # V5 _check_bullet_discipline, # V8 _check_ship_to_required, # V6 _check_address_shape, # V7 _check_required_fields, # V9 _check_po_identity, # V10 _check_sentinels, # V11 _check_hygiene, # V12 _check_qty_unit_price, # V13 ) def _validate_new_po_values(candidate, email_data): """Value-level gate rules V1-V13 for coupa_new_po. FAIL CLOSED. V4 anchor integrity builds the shared anchor frame first (every later byte proof needs it); the remaining rules then run in spec order via the per-rule _check_* helpers, each re-deriving its evidence from email_data["body"] so an extractor bug cannot vouch for itself. The first helper to return a reason code short-circuits to (False, reason).""" lines = _plain_lines(email_data["body"]) items = candidate.get("line_items") or [] frame, reason = _build_anchor_frame(lines, items) if reason: return False, reason for check in _NEW_PO_VALUE_CHECKS: reason = check(candidate, frame) if reason: return False, reason return True, "ok" def try_deterministic_parse(email_data): """Entry point. Returns (parsed|None, parse_method, template_id, reason). On a proven-conformant parse returns (dict, 'template', template_id, 'ok'). On any miss/invalid/exception returns (None, 'ai_fallback', template_id, reason) -- a failure is NEVER a parsed result.""" template_id = "unknown" try: template_id, reason = classify_template(email_data) if template_id == "unknown": return None, "ai_fallback", template_id, reason if template_id == "coupa_new_po": candidate = extract_new_po(email_data) else: candidate = extract_cancellation(email_data) ok, reason = validate(candidate, template_id, email_data) if not ok: return None, "ai_fallback", template_id, reason return candidate, "template", template_id, "ok" except Exception: # noqa: BLE001 -- fail closed on ANY extractor error return None, "ai_fallback", template_id, "extractor_raised" # --------------------------------------------------------------------------- # AI-fallback validation gate -- FAIL CLOSED # # Called on the raw Bedrock/Claude output BEFORE enrich_parsed and BEFORE any # dispatch/save (handler.py). Mirrors WO's validate_ai_fallback, but PO's # contract is NESTED and requires missing-key normalization, so the gate returns # a THREE-tuple (ok, reason, normalized_candidate_or_None): on success the # handler adopts `parsed = normalized` and never re-normalizes. # # Missing keys are TOLERATED (the LLM may omit null fields) and filled with None # at every nesting level; EXTRA keys are REJECTED with "key_set_mismatch" at # every nesting level. This deliberately does NOT reuse _normalize(), which # silently drops extras and coerces line_items [] -> [one empty item] (that # would change the downstream write shape -- the AI_PAYLOAD fixture ships # line_items: [] and it must stay []). # --------------------------------------------------------------------------- # Top-level scalar fields that must be None or str (blocks LLM-emitted maps/ # lists from landing as DynamoDB Map/List attribute pollution). email_type, # po_number, po_status are validated separately; supplier/ship_to/line_items are # nested; total_amount is a money field. _AI_TOP_STR_FIELDS = ( "source_system", "submitted_by", "on_behalf_of", "order_date", "revision_date", "last_opened", "acknowledged_at", "payment_terms", "requisition_number", "department", "view_order_url", "site_code", "currency", "fiscal_year", "trade", "coupa_category", ) # Line-item scalar fields that must be None or str. amount is a money field; # quantity/price are money-or-str (enrich_parsed coerces numeric strings). _AI_LINE_ITEM_STR_FIELDS = ( "description", "currency", "need_by", "category", "account_code", "period", "unit", ) def _is_ai_money(value): """True for a valid strict money value: None | int | Decimal. PO parses Bedrock output with parse_float=Decimal, so a float can never legitimately occur and a float-typed check would be wrong. bool is an int subclass and is EXPLICITLY rejected (a JSON true/false must not read as 1/0 into a money column).""" if value is None: return True if isinstance(value, bool): return False return isinstance(value, (int, Decimal)) def _is_ai_money_or_str(value): """True for None | int | Decimal | str, bool rejected. str is tolerated for quantity/price because enrich_parsed's shared coercion stage converts numeric strings to Decimal and deliberately stores non-numeric strings verbatim -- the gate must not break that documented contract.""" if isinstance(value, str): return True return _is_ai_money(value) def _normalize_nested_dict(value, keys): """Strict per-level normalize for a nested container (supplier/ship_to). Returns (normalized_dict_or_None, ok): * None -> ({k: None for k in keys}, True) (all-None dict) * dict whose keys are a subset of `keys` -> (missing filled None, True) * dict with any EXTRA key -> (None, False) * any other type -> (None, False) """ if value is None: return {k: None for k in keys}, True if not isinstance(value, dict): return None, False if set(value.keys()) - set(keys): return None, False return {k: value.get(k) for k in keys}, True def validate_ai_fallback(candidate): # noqa: C901, PLR0911, PLR0912 """Fail-closed schema/type validation for the AI-fallback parse path. Returns (ok, reason, normalized_candidate_or_None). On success the handler adopts the returned normalized dict (`parsed = normalized`) and never re-normalizes. Reason-code vocabulary: not_an_object, key_set_mismatch, missing_required_field, invalid_status, invalid_money_type, invalid_field_type, ok.""" # (1) json.loads on model output can yield list/str/int/None; only an object # can satisfy the contract. Anything else must fail closed HERE rather than # AttributeError at the handler's logger f-string into async retries / DLQ. if not isinstance(candidate, dict): return False, "not_an_object", None # (2) key-set + missing-key normalization: extras rejected, missing -> None. if set(candidate.keys()) - set(CONTRACT_KEYS): return False, "key_set_mismatch", None normalized = {k: candidate.get(k) for k in CONTRACT_KEYS} supplier, ok = _normalize_nested_dict(normalized["supplier"], SUPPLIER_KEYS) if not ok: return False, "key_set_mismatch", None normalized["supplier"] = supplier ship_to, ok = _normalize_nested_dict(normalized["ship_to"], SHIP_TO_KEYS) if not ok: return False, "key_set_mismatch", None normalized["ship_to"] = ship_to # line_items: list or None. None -> []; [] stays [] (preserves the current # downstream write shape). Every element must be a dict; each is normalized # to exactly LINE_ITEM_KEYS with extras rejected. items = normalized["line_items"] if items is None: items = [] elif not isinstance(items, list): return False, "key_set_mismatch", None norm_items = [] for it in items: if not isinstance(it, dict): return False, "key_set_mismatch", None if set(it.keys()) - set(LINE_ITEM_KEYS): return False, "key_set_mismatch", None norm_items.append({k: it.get(k) for k in LINE_ITEM_KEYS}) normalized["line_items"] = norm_items # (3) po_number: required non-empty, hardened prefix+hyphen+digits shape. po = normalized["po_number"] if not po or not _AI_PO_ID_RE.match(str(po)): return False, "missing_required_field", None # (4) email_type in the enum, enforced HERE (before dispatch) so a miss can # never fall into the handler's else -> save_new_po branch. isinstance guard # first: an unhashable JSON list/dict would raise TypeError on `in ` # and escape the fail-closed gate. et = normalized["email_type"] if not isinstance(et, str) or et not in VALID_EMAIL_TYPES: return False, "missing_required_field", None # (5) po_status: None or str. PARTIAL DIVERGENCE from WO -- PO has NO closed # AI-path status vocabulary (NEW_PO_SAFE_STATUSES is a template-path new_po # allow-list; revision/cancellation statuses are uncharacterized), so # arbitrary strings pass the type check -- with ONE exception: a # non-cancellation email_type may not carry the sticky "Cancelled" status. # Dispatch routes on email_type, so an AI-path new_po/revision carrying # po_status="Cancelled" would reach save_new_po/save_revision and cancel a # live PO via _merge_update while never hitting save_cancellation. The # template path already forbids this (a cancellation misrouted as new_po # defeats the sticky-Cancelled guard); mirror it here. email_type is already # validated to the enum at step (4); a cancellation reaches save_cancellation, # which hardcodes the status, so po_status is irrelevant on that route. status = normalized["po_status"] if status is not None and not isinstance(status, str): return False, "invalid_status", None if status == _CANCELLED_STATUS and normalized["email_type"] != "cancellation": return False, "invalid_status", None # (6) money fields, two tiers. if not _is_ai_money(normalized["total_amount"]): return False, "invalid_money_type", None for it in normalized["line_items"]: if not _is_ai_money(it["amount"]): return False, "invalid_money_type", None if not _is_ai_money_or_str(it["quantity"]): return False, "invalid_money_type", None if not _is_ai_money_or_str(it["price"]): return False, "invalid_money_type", None # (7) all remaining scalar fields must be None or str. for field in _AI_TOP_STR_FIELDS: val = normalized[field] if val is not None and not isinstance(val, str): return False, "invalid_field_type", None if normalized["supplier"]["name"] is not None and not isinstance( normalized["supplier"]["name"], str ): return False, "invalid_field_type", None for field in SHIP_TO_KEYS: val = normalized["ship_to"][field] if val is not None and not isinstance(val, str): return False, "invalid_field_type", None for it in normalized["line_items"]: for field in _AI_LINE_ITEM_STR_FIELDS: val = it[field] if val is not None and not isinstance(val, str): return False, "invalid_field_type", None return True, "ok", normalized