"""Deterministic derived-field classifiers for Coupa purchase order emails. Pure module: stdlib only -- no boto3, no network, no imports from handler or template_parser. Computes the three DERIVED_KEYS (site_code, fiscal_year, trade) that the template parser and the LLM path both leave for a shared post-stage to fill identically (see template_parser.py DERIVED_KEYS / enrich_parsed). This is a FAITHFUL v1 port of the English rules in the handler's EXTRACTION_PROMPT sections "## site_code extraction", "## fiscal_year" and "## trade classification". A corpus backtest will harden the thresholds later; this module intentionally invents no rule beyond those three sections. TOTALITY (hard requirement): this runs on the untrusted email path in an S3-async Lambda, where an uncaught exception means infinite retry -> DLQ -> silent data loss. Every public function therefore mirrors template_parser.try_deterministic_parse: it wraps its whole body in a broad try/except and NEVER raises -- a bad/hostile input yields None (or an all-None dict for derive_all), never a traceback. Inputs are treated as hostile: wrong types, None, non-dict line items and multi-megabyte strings are all tolerated. Scans are length-capped and every regex is linear (no nested quantifiers, no catastrophic backtracking). Public API: derive_site_code(parsed) -> str | None derive_fiscal_year(parsed) -> str | None derive_trade(parsed) -> str | None derive_all(parsed) -> {"site_code":..., "trade":..., "fiscal_year":...} `parsed` is the post-extraction contract dict from either path. Relevant inputs: parsed["ship_to"]["name"], parsed["ship_to"]["attn"], parsed["line_items"][*]["description"|"amount"|"need_by"], parsed["order_date"]. """ import re from decimal import Decimal, InvalidOperation # --- scan caps (hostile-input blast-radius limits) --------------------------- _MAX_ITEMS = 500 # line_items scanned at most _MAX_NAME = 4096 # ship_to name/attn chars scanned for a site code _MAX_DESC = 4096 # description chars scanned for a bracketed/standalone code _MAX_DATE = 1024 # date-string chars scanned for a year _MAX_TRADE = 100_000 # description chars scanned for trade keywords (per item) _MAX_TRADE_TOTAL = 200_000 # aggregate description chars classified per PO # --------------------------------------------------------------------------- # site_code # --------------------------------------------------------------------------- # Token shape: 3-5 chars, uppercase A-Z0-9, MUST start with a letter. Codes may # be all letters (e.g. KLAL) -- a digit is NOT required. _STRICT_CODE_RE = re.compile(r"[A-Z][A-Z0-9]{2,4}") # Same shape, but only as a standalone token (not embedded in a longer # alphanumeric run): '–WKY3' matches, 'Station' does not, 'DLI6X99' does not. _CODE_TOKEN_RE = re.compile(r"(?=2 alphabetic characters. Measured over all 1,063 distinct LLM-extracted codes: 999 have 3 letters, 7 have 4, 3 have 5, and NONE have fewer than 2. Sub-2-letter tokens (e.g. 'B187') are PO-prefix-style garbage, so every site-code shape rejects them. """ return tok not in _SKIP and sum(c.isalpha() for c in tok) >= 2 def _last_valid_code(text): """Rightmost valid code-shaped token in `text`. This is the multi-hop resolver: an ATTN like 'CBRE - RME - DLI6' yields the rightmost code that passes _is_valid_code (DLI6), skipping the RME hop. """ for tok in reversed(_CODE_TOKEN_RE.findall(text)): if _is_valid_code(tok): return tok return None def _code_name_leading(name): """Leading shape (highest priority): the name OPENS with a standalone code-shaped token containing >=1 digit, delimited by space/dash/'(' -- 'WPT2 - Amazon...' -> WPT2, 'QDE1 (Co-located inside WDE1)' -> QDE1. The digit requirement + leading position outrank the parens shape, so a 'QDE1 (... WDE1)' resolves to the opener QDE1, not the host building WDE1. """ if not isinstance(name, str): return None m = _LEADING_CODE_RE.match(name.strip()[:_MAX_NAME]) if not m: return None tok = m.group(1) if _is_valid_code(tok) and any(c.isdigit() for c in tok): return tok return None def _code_name_parens(name): """Shape 1: ship-to name in parentheses -- 'Services LLC (KLAL)' -> KLAL. Uses the rightmost-non-skip resolver (same as ATTN) so parens content like '(ATTN: Wagon Wheel DS Station -WKY3)' resolves to WKY3, not the skip-listed ATTN hop. """ if not isinstance(name, str): return None for content in _PAREN_RE.findall(name[:_MAX_NAME]): code = _last_valid_code(content) if code: return code return None def _code_name_after_dash(name): """Shape 2: ship-to name with a dash. Real ship-to names carry the code on EITHER side of the dash -- 'Amazon... LLC - SNY5' (after) and 'WPT2 - Amazon... LLC' (before). First prefer any segment that IS exactly a code-shaped token (leftmost such segment wins); only if no whole segment is an exact code, fall back to scanning the tail after the last dash (so a code-shaped word before the dash, e.g. 'LLC', can never leak in when the real after-dash code is skip-listed). """ if not isinstance(name, str): return None s = name[:_MAX_NAME] idx = max(s.rfind(d) for d in _DASHES) if idx == -1: return None for seg in _DASH_SPLIT_RE.split(s): tok = seg.strip() if _STRICT_CODE_RE.fullmatch(tok) and _is_valid_code(tok): return tok return _last_valid_code(s[idx + 1 :]) def _code_name_midtoken(name): """Mid-name shape (lower priority): a standalone uppercase code-shaped token with >=1 digit anywhere in the name -- 'Amazon Fresh UVA5 Non-Inv (Prime)' -> UVA5. The digit requirement means all-letter words (FRESH, ...) can never match; mixed-case words never match the uppercase token shape at all. """ if not isinstance(name, str): return None for tok in _CODE_TOKEN_RE.findall(name[:_MAX_NAME]): if _is_valid_code(tok) and any(c.isdigit() for c in tok): return tok return None def _code_from_attn(attn): """Shapes 3 & 4: ship-to ATTN line (with dash/en-dash, or direct). Unified: the rightmost non-skip code-shaped token across the whole ATTN value. Covers 'Wagon Wheel DS Station -WKY3' -> WKY3, 'HJX1' -> HJX1, and the multi-hop 'CBRE - RME - DLI6' -> DLI6. """ if not isinstance(attn, str): return None return _last_valid_code(attn[:_MAX_NAME]) def _code_name_is(name): """Shape 5: the ship-to name IS the code -- 'DBU2' -> DBU2.""" if not isinstance(name, str): return None n = name.strip() if _STRICT_CODE_RE.fullmatch(n) and _is_valid_code(n): return n return None def _code_desc_prefix(desc): """Shape 6: line-item description prefix -- 'DYO1 - ... - ...' -> DYO1.""" if not isinstance(desc, str): return None m = _PREFIX_CODE_RE.match(desc[:80]) if m and _is_valid_code(m.group(1)): return m.group(1) return None def _code_desc_bracket(desc): """Shape 7: line-item description in brackets -- '[HMK4] ...' -> HMK4.""" if not isinstance(desc, str): return None for m in _BRACKET_CODE_RE.finditer(desc[:_MAX_DESC]): if _is_valid_code(m.group(1)): return m.group(1) return None def _code_desc_freetoken(desc): """Shape 8 (lowest priority): a standalone uppercase token, length 4-5, letter-first, with >=1 digit, anywhere in a description -- 'Need Sea Haven to pump out waste water aqt DSF7' -> DSF7. The length-4 floor (vs the 3-char floor of other shapes) and the digit requirement keep this loose free-scan from grabbing 3-letter uppercase words or all-letter tokens mid-sentence. """ if not isinstance(desc, str): return None for tok in _FREE_CODE_TOKEN_RE.findall(desc[:_MAX_DESC]): if _is_valid_code(tok) and any(c.isdigit() for c in tok): return tok return None def _iter_items(parsed): """Yield up to _MAX_ITEMS dict line items from parsed; tolerant of junk.""" items = parsed.get("line_items") if not isinstance(items, list): return for item in items[:_MAX_ITEMS]: if isinstance(item, dict): yield item def derive_site_code(parsed): """Derive the Amazon facility site_code, or None. NEVER raises. Shapes are tried in strict priority order. A skip-listed sole candidate for a shape yields None for that shape and evaluation continues to the next: leading -- name opens with a digit-bearing code ('WPT2 - ...') 1 parens -- code inside '(...)' 2 dash -- exact code-shaped segment either side of a dash 3/4 attn -- rightmost non-skip code in the ATTN line 5 name-is-- the name IS the code midtoken -- a digit-bearing code anywhere in the name 6 prefix -- description leading code 7 bracket-- description '[CODE]' 8 free -- a length 4-5 digit-bearing code anywhere in a description """ try: if not isinstance(parsed, dict): return None ship_to = parsed.get("ship_to") if not isinstance(ship_to, dict): ship_to = {} name = ship_to.get("name") attn = ship_to.get("attn") for shape in ( _code_name_leading(name), # leading (highest) _code_name_parens(name), # 1 _code_name_after_dash(name), # 2 _code_from_attn(attn), # 3 & 4 _code_name_is(name), # 5 _code_name_midtoken(name), # mid-name ): if shape: return shape for item in _iter_items(parsed): # 6 code = _code_desc_prefix(item.get("description")) if code: return code for item in _iter_items(parsed): # 7 code = _code_desc_bracket(item.get("description")) if code: return code for item in _iter_items(parsed): # 8 code = _code_desc_freetoken(item.get("description")) if code: return code return None except Exception: # noqa: BLE001 -- totality: never raise on the email path return None # --------------------------------------------------------------------------- # fiscal_year # --------------------------------------------------------------------------- # A standalone 4-digit 20xx year. Word-bounded so it never matches digits buried # inside a longer run (e.g. the '2062' inside a PO id '18206023'). _YEAR_RE = re.compile(r"\b(20\d{2})\b") # MM/DD/YY -> the trailing 2-digit year is expanded to 20YY. _SHORT_DATE_RE = re.compile(r"\b\d{1,2}/\d{1,2}/(\d{2})\b") def _year_from_date(value): """Extract a 4-digit year from a date string: a literal 20xx, else a MM/DD/YY whose YY expands to 20YY. Returns the 4-digit string or None.""" if not isinstance(value, str): return None s = value[:_MAX_DATE] m = _YEAR_RE.search(s) if m: return m.group(1) m = _SHORT_DATE_RE.search(s) if m: return "20" + m.group(1) return None def _year_from_text(value): """A standalone 20xx year anywhere in free text, or None.""" if not isinstance(value, str): return None m = _YEAR_RE.search(value[:_MAX_DESC]) return m.group(1) if m else None def derive_fiscal_year(parsed): """Derive the 4-digit fiscal_year string, or None. NEVER raises. Fallback order: (1) order_date year, (2) any line-item need_by year, (3) a standalone 20xx year in any line-item description. """ try: if not isinstance(parsed, dict): return None year = _year_from_date(parsed.get("order_date")) if year: return year for item in _iter_items(parsed): year = _year_from_date(item.get("need_by")) if year: return year for item in _iter_items(parsed): year = _year_from_text(item.get("description")) if year: return year return None except Exception: # noqa: BLE001 -- totality: never raise on the email path return None # --------------------------------------------------------------------------- # trade # --------------------------------------------------------------------------- # Keyword matching interpretation (judgment call, documented): each keyword is # matched with a LEADING word boundary and no trailing boundary -- i.e. it hits # the keyword as a whole word OR as the prefix of a longer word. This makes # 'sign' match 'signage', 'dock door' match 'dock doors', and 'roof' match # 'roofing' (desired), while NOT matching a keyword buried mid-word ('ice' in # 'service'/'price', 'lock' in 'block'/'clock', 'gate' in 'mitigate') -- the # catastrophic false positives a bare substring test would produce. All matching # is done on a lower-cased, length-capped copy of the description. def _kw(*words): """Compile a linear leading-boundary alternation of literal keywords.""" body = "|".join(re.escape(w) for w in words) return re.compile(r"\b(?:" + body + r")") _PM_RE = _kw( "plumbing pm", "plumbing preventative", "plumbing maintenance", "plumbing - backflow", "plumbing - water heater", ) _PLUMBING_RE = _kw("plumbing") # BBM-structured 'Plumbing - - - BBM' row: 'plumbing' opening # the description immediately followed by a dash. Resolved by a measured ladder # (see _classify_desc) that matches the LLM baseline over the full corpus. _BBM_PLUMBING_RE = re.compile(r"plumbing\s*[-–—]") # Ladder rungs for a BBM plumbing row, checked in order: # (a) emergency/reactive -> Reactive _BBM_PLUMBING_REACTIVE_WORD_RE = _kw("reactive", "emergency") # (b) technician -> PM _BBM_PLUMBING_TECH_RE = _kw("technician") # (c) water heater / backflow / water fountain -> PM (before (d), so a # 'water heater repair' hits PM here and not Reactive on 'repair') _BBM_PLUMBING_PM_RE = _kw("water heater", "backflow", "water fountain") # (d) clog/unclog/leak/repair/sewer/drain -> Reactive ('clog'/'leak' prefixes # also catch 'clogged'/'leaking') _BBM_PLUMBING_REACTIVE_RE = _kw("clog", "unclog", "leak", "repair", "sewer", "drain") # (e) else (project, install, misc) -> PM # Plumbing-specific qualifiers -- these indicate Plumbing - Reactive on their # OWN, without the word 'plumbing' present ('CLOGGED PIT AUGER' -> Reactive). _PLUMBING_SPECIFIC_RE = _kw( "clog", "unclog", "sewer", "drain", "grease trap", "jetter", "toilet", "faucet", "urinal", ) # Generic reactive qualifiers -- these require the word 'plumbing' to be present # before they route to Plumbing - Reactive. _PLUMBING_GENERIC_RE = _kw( "reactive", "emergency", "repair", "leak", "flood", "water line", "pipe", ) _ELECTRICAL_RE = _kw( "electrical", "lighting", "ballast", "outlet", "circuit", "panel", "generator", "transformer", "conduit", ) _HVAC_RE = _kw( "hvac", "heating", "cooling", "air conditioning", "rtu", "ahu", "vav", "refrigerant", "thermostat", "ductwork", ) _DOCK_DOORS_RE = _kw( "dock door", "dock leveler", "dock plate", "dock seal", "dock bumper", ) _DOORS_RE = _kw("door", "overhead door", "roll-up", "automatic door", "access door") _SIGNAGE_RE = _kw("sign", "banner", "wayfinding", "marquee", "directional") _CARPENTRY_RE = _kw("carpentry", "cabinet", "millwork", "trim", "shelving", "framing") _FENCE_RE = _kw("fence", "fencing", "bollard") _GATE_RE = _kw("gate") _CONVEYANCE_RE = _kw("conveyor", "conveyance", "mhe", "material handling", "sortation") _PAINTING_RE = _kw("paint", "painting", "primer", "coating", "touch-up") _FLOORING_RE = _kw("floor", "tile", "carpet", "epoxy", "polishing") _JANITORIAL_RE = _kw( "janitorial", "cleaning", "custodial", "pressure wash", "power wash" ) _FIRE_RE = _kw("fire", "sprinkler", "extinguisher", "fire alarm", "suppression") _LANDSCAPING_RE = _kw("landscape", "lawn", "tree", "yard", "mowing", "irrigation") _ROOFING_RE = _kw("roof", "roofing", "gutter", "downspout") # Security/Locksmith: 'lock' and 'key' match as EXACT whole words only (so # 'Locker' is not 'lock' and 'keyboard' is not 'key'); the multi-word terms keep # the leading-boundary/prefix behavior. The phrase 'key box' is stripped before # this test (see _classify_desc) so a Key Box fixture never triggers Locksmith. _SECURITY_RE = re.compile(r"\b(?:lock|key)\b|\b(?:access control|camera|security|cctv)") _SNOW_RE = _kw("snow", "ice", "salt", "de-ice", "plow") _EMERGENCY_RE = _kw("emergency") # General Building catch-all sub-shapes. _HANDYMAN_RE = re.compile(r"\bhandyman\b") _PROJECT_RE = re.compile(r"\bproject\b") TRADE_UPLIFT = "PO Uplift" TRADE_GENERAL_BUILDING = "General Building" # Simple (single-regex) trades, in strict priority order. The compound and # exception-bearing trades (Plumbing, Electrical, Fencing/Gates, the General # Building family, PO Uplift) are handled inline in _classify_desc. _SIMPLE_TRADES = ( (_HVAC_RE, "HVAC"), (_DOCK_DOORS_RE, "Dock Doors"), (_DOORS_RE, "Doors"), (_SIGNAGE_RE, "Signage"), (_CARPENTRY_RE, "Carpentry"), ) # Simple trades that follow Fencing/Gates in the priority table. Security and # Snow are handled explicitly after this loop (Security needs the 'key box' # exclusion applied to its match target, so it can't share the generic loop). _SIMPLE_TRADES_TAIL = ( (_CONVEYANCE_RE, "Conveyance/MHE"), (_PAINTING_RE, "Painting"), (_FLOORING_RE, "Flooring"), (_JANITORIAL_RE, "Janitorial"), (_FIRE_RE, "Fire/Life Safety"), (_LANDSCAPING_RE, "Landscaping/Yard"), (_ROOFING_RE, "Roofing"), ) def _classify_desc(desc): """Classify a single non-empty description into one trade label. Returns a label for any non-empty description -- 'General Building' is the catch-all when nothing else matches. Callers must not pass empty/blank descriptions (that case is handled upstream so it can return None). """ low = desc[:_MAX_TRADE].lower() stripped = low.strip() # PO Uplift: exactly or primarily "PO Uplift". if stripped == "po uplift" or stripped.startswith("po uplift"): return TRADE_UPLIFT # BBM-structured 'Plumbing - ...' row: resolved by a measured ladder tuned to # the LLM baseline over the full corpus (first rung wins). if _BBM_PLUMBING_RE.match(stripped): if _BBM_PLUMBING_REACTIVE_WORD_RE.search(low): # (a) emergency/reactive return "Plumbing - Reactive" if _BBM_PLUMBING_TECH_RE.search(low): # (b) technician return "Plumbing - PM" if _BBM_PLUMBING_PM_RE.search(low): # (c) heater/backflow/fountain return "Plumbing - PM" if _BBM_PLUMBING_REACTIVE_RE.search(low): # (d) clog/leak/repair/... return "Plumbing - Reactive" return "Plumbing - PM" # (e) project/install/misc # Plumbing - PM (before Reactive) for non-BBM free text. if _PM_RE.search(low): return "Plumbing - PM" # Plumbing - Reactive: a plumbing-specific qualifier ALONE (clog, drain, # toilet, ...), or the word 'plumbing' PLUS a generic reactive qualifier. if _PLUMBING_SPECIFIC_RE.search(low): return "Plumbing - Reactive" if _PLUMBING_RE.search(low) and _PLUMBING_GENERIC_RE.search(low): return "Plumbing - Reactive" # Electrical -- but electrical keywords do NOT match in a dock-door context. if _ELECTRICAL_RE.search(low) and "dock door" not in low: return "Electrical" for regex, label in _SIMPLE_TRADES: if regex.search(low): return label # Fencing/Gates: fence/fencing/bollard anywhere, or 'gate' but NOT 'dock # gate' (the dock-gate occurrences are removed before the gate test). if _FENCE_RE.search(low) or _GATE_RE.search(low.replace("dock gate", " ")): return "Fencing/Gates" for regex, label in _SIMPLE_TRADES_TAIL: if regex.search(low): return label # Security/Locksmith: 'lock'/'key' as whole words, with 'key box' stripped # first (mirrors the 'dock gate' exclusion) so a Key Box fixture is not one. if _SECURITY_RE.search(low.replace("key box", " ")): return "Security/Locksmith" if _SNOW_RE.search(low): return "Snow Removal" # General Building family (catch-alls, in priority order). if stripped.startswith("emer") or _EMERGENCY_RE.search(low): return "General Building - Emergency" if _HANDYMAN_RE.search(low) or "general building technician" in low: return "General Building - Handyman" if "general building project" in low or _PROJECT_RE.search(low): return "General Building - Project" # Keyword-less BBM 'General Building - - ...' rows (parking lot, # fan, ceiling, television, locker, key box, ...) fall through here to the # plain catch-all -- that matches the measured LLM majority for such rows. return TRADE_GENERAL_BUILDING def _to_amount(value): """Coerce a line amount (Decimal | str | int/float | junk) to a Decimal for comparison. Anything unparseable becomes Decimal(0) -- defensive, never raises, so an item with a bad amount simply cannot win the highest-value tie-break.""" if isinstance(value, bool): return Decimal(0) if isinstance(value, Decimal): return value if value.is_finite() else Decimal(0) if isinstance(value, int): return Decimal(value) if isinstance(value, float): try: d = Decimal(str(value)) return d if d.is_finite() else Decimal(0) except InvalidOperation: return Decimal(0) if isinstance(value, str): try: d = Decimal(value.replace(",", "").strip()) return d if d.is_finite() else Decimal(0) except (InvalidOperation, ValueError): return Decimal(0) return Decimal(0) def derive_trade(parsed): """Derive the PO's primary trade label, or None. NEVER raises. Each line item with a usable description is classified. The PO trade is the primary non-"PO Uplift" trade; when several distinct non-uplift trades are present, the trade of the highest-`amount` non-uplift item wins. If every described item is PO Uplift, returns "PO Uplift". With no line items or no usable descriptions at all, returns None ("General Building" is only ever returned for a description that matched nothing). """ try: if not isinstance(parsed, dict): return None classified = [] # (trade_label, amount_decimal) # Aggregate CPU budget across ALL items: the per-item _MAX_TRADE cap # alone still allows _MAX_ITEMS x _MAX_TRADE x ~25 regex passes # (~10s full-core, minutes under the 256MB Lambda's CPU throttle -> # timeout -> async retry -> DLQ). Real POs total well under 10k chars # of description text; stop classifying once the budget is spent and # decide from what was classified (sh-security-review # PO-DERIVED-AVAIL-001 defense-in-depth). budget = _MAX_TRADE_TOTAL for item in _iter_items(parsed): desc = item.get("description") if not isinstance(desc, str) or not desc.strip(): continue if budget <= 0: break desc = desc[:budget] budget -= len(desc) classified.append((_classify_desc(desc), _to_amount(item.get("amount")))) if not classified: return None non_uplift = [(t, a) for (t, a) in classified if t != TRADE_UPLIFT] if not non_uplift: return TRADE_UPLIFT if len({t for (t, _) in non_uplift}) == 1: return non_uplift[0][0] best_trade, best_amount = non_uplift[0] for trade, amount in non_uplift[1:]: if amount > best_amount: best_trade, best_amount = trade, amount return best_trade except Exception: # noqa: BLE001 -- totality: never raise on the email path return None def derive_all(parsed): """All three derived fields as a dict. NEVER raises: on any failure returns an all-None dict so the caller always gets the three keys.""" try: return { "site_code": derive_site_code(parsed), "trade": derive_trade(parsed), "fiscal_year": derive_fiscal_year(parsed), } except Exception: # noqa: BLE001 -- totality: never raise on the email path return {"site_code": None, "trade": None, "fiscal_year": None}