mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-10-01 12:23:13 +00:00
677 lines
26 KiB
Python
677 lines
26 KiB
Python
|
|
"""Deterministic derived-field classifiers for Coupa purchase order emails.
|
|||
|
|
|
|||
|
|
Pure module: stdlib only -- no boto3, no network, no imports from handler or
|
|||
|
|
template_parser. Computes the three DERIVED_KEYS (site_code, fiscal_year, trade)
|
|||
|
|
that the template parser and the LLM path both leave for a shared post-stage to
|
|||
|
|
fill identically (see template_parser.py DERIVED_KEYS / enrich_parsed).
|
|||
|
|
|
|||
|
|
This is a FAITHFUL v1 port of the English rules in the handler's
|
|||
|
|
EXTRACTION_PROMPT sections "## site_code extraction", "## fiscal_year" and
|
|||
|
|
"## trade classification". A corpus backtest will harden the thresholds later;
|
|||
|
|
this module intentionally invents no rule beyond those three sections.
|
|||
|
|
|
|||
|
|
TOTALITY (hard requirement): this runs on the untrusted email path in an
|
|||
|
|
S3-async Lambda, where an uncaught exception means infinite retry -> DLQ ->
|
|||
|
|
silent data loss. Every public function therefore mirrors
|
|||
|
|
template_parser.try_deterministic_parse: it wraps its whole body in a broad
|
|||
|
|
try/except and NEVER raises -- a bad/hostile input yields None (or an all-None
|
|||
|
|
dict for derive_all), never a traceback. Inputs are treated as hostile: wrong
|
|||
|
|
types, None, non-dict line items and multi-megabyte strings are all tolerated.
|
|||
|
|
Scans are length-capped and every regex is linear (no nested quantifiers, no
|
|||
|
|
catastrophic backtracking).
|
|||
|
|
|
|||
|
|
Public API:
|
|||
|
|
derive_site_code(parsed) -> str | None
|
|||
|
|
derive_fiscal_year(parsed) -> str | None
|
|||
|
|
derive_trade(parsed) -> str | None
|
|||
|
|
derive_all(parsed) -> {"site_code":..., "trade":..., "fiscal_year":...}
|
|||
|
|
|
|||
|
|
`parsed` is the post-extraction contract dict from either path. Relevant inputs:
|
|||
|
|
parsed["ship_to"]["name"], parsed["ship_to"]["attn"],
|
|||
|
|
parsed["line_items"][*]["description"|"amount"|"need_by"], parsed["order_date"].
|
|||
|
|
"""
|
|||
|
|
|
|||
|
|
import re
|
|||
|
|
from decimal import Decimal, InvalidOperation
|
|||
|
|
|
|||
|
|
# --- scan caps (hostile-input blast-radius limits) ---------------------------
|
|||
|
|
_MAX_ITEMS = 500 # line_items scanned at most
|
|||
|
|
_MAX_NAME = 4096 # ship_to name/attn chars scanned for a site code
|
|||
|
|
_MAX_DESC = 4096 # description chars scanned for a bracketed/standalone code
|
|||
|
|
_MAX_DATE = 1024 # date-string chars scanned for a year
|
|||
|
|
_MAX_TRADE = 100_000 # description chars scanned for trade keywords (per item)
|
|||
|
|
_MAX_TRADE_TOTAL = 200_000 # aggregate description chars classified per PO
|
|||
|
|
|
|||
|
|
# ---------------------------------------------------------------------------
|
|||
|
|
# site_code
|
|||
|
|
# ---------------------------------------------------------------------------
|
|||
|
|
# Token shape: 3-5 chars, uppercase A-Z0-9, MUST start with a letter. Codes may
|
|||
|
|
# be all letters (e.g. KLAL) -- a digit is NOT required.
|
|||
|
|
_STRICT_CODE_RE = re.compile(r"[A-Z][A-Z0-9]{2,4}")
|
|||
|
|
# Same shape, but only as a standalone token (not embedded in a longer
|
|||
|
|
# alphanumeric run): '–WKY3' matches, 'Station' does not, 'DLI6X99' does not.
|
|||
|
|
_CODE_TOKEN_RE = re.compile(r"(?<![A-Za-z0-9])[A-Z][A-Z0-9]{2,4}(?![A-Za-z0-9])")
|
|||
|
|
_PAREN_RE = re.compile(r"\(([^()\n]{1,40})\)")
|
|||
|
|
_BRACKET_CODE_RE = re.compile(r"\[([A-Z][A-Z0-9]{2,4})\]")
|
|||
|
|
# Description prefix: leading code then a dash/en-dash/em-dash separator.
|
|||
|
|
_PREFIX_CODE_RE = re.compile(r"\s*([A-Z][A-Z0-9]{2,4})\s*[-–—]")
|
|||
|
|
# Leading code-shaped token that opens the ship-to name and is delimited by a
|
|||
|
|
# space, dash or '(' -- e.g. 'WPT2 - Amazon...' / 'QDE1 (Co-located ...)'.
|
|||
|
|
_LEADING_CODE_RE = re.compile(r"([A-Z][A-Z0-9]{2,4})(?=[\s\-–—(])")
|
|||
|
|
# Free-token code in a description: standalone, length 4-5, letter-first.
|
|||
|
|
_FREE_CODE_TOKEN_RE = re.compile(r"(?<![A-Za-z0-9])[A-Z][A-Z0-9]{3,4}(?![A-Za-z0-9])")
|
|||
|
|
_DASHES = "-–—"
|
|||
|
|
_DASH_SPLIT_RE = re.compile(r"[-–—]")
|
|||
|
|
|
|||
|
|
# Never a site_code (from the prompt's "Not site codes" list). If the only
|
|||
|
|
# candidate for a shape is skip-listed, that shape yields None and evaluation
|
|||
|
|
# falls through to the next lower-priority shape.
|
|||
|
|
_SKIP = frozenset(
|
|||
|
|
{
|
|||
|
|
"RME",
|
|||
|
|
"BBM",
|
|||
|
|
"JLL",
|
|||
|
|
"PARAG",
|
|||
|
|
"ERIK",
|
|||
|
|
"HVAC",
|
|||
|
|
"LED",
|
|||
|
|
"PVC",
|
|||
|
|
"ADA",
|
|||
|
|
"OSHA",
|
|||
|
|
"EMR",
|
|||
|
|
"BMS",
|
|||
|
|
"DDC",
|
|||
|
|
"MRO",
|
|||
|
|
"NTE",
|
|||
|
|
"EST",
|
|||
|
|
# Corporate-suffix / addressing tokens -- never a site code.
|
|||
|
|
"LLC",
|
|||
|
|
"INC",
|
|||
|
|
"CORP",
|
|||
|
|
"LTD",
|
|||
|
|
"ATTN",
|
|||
|
|
}
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _is_valid_code(tok):
|
|||
|
|
"""A code-shaped token is a real site code only if it is not skip-listed AND
|
|||
|
|
carries >=2 alphabetic characters.
|
|||
|
|
|
|||
|
|
Measured over all 1,063 distinct LLM-extracted codes: 999 have 3 letters, 7
|
|||
|
|
have 4, 3 have 5, and NONE have fewer than 2. Sub-2-letter tokens (e.g.
|
|||
|
|
'B187') are PO-prefix-style garbage, so every site-code shape rejects them.
|
|||
|
|
"""
|
|||
|
|
return tok not in _SKIP and sum(c.isalpha() for c in tok) >= 2
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _last_valid_code(text):
|
|||
|
|
"""Rightmost valid code-shaped token in `text`.
|
|||
|
|
|
|||
|
|
This is the multi-hop resolver: an ATTN like 'CBRE - RME - DLI6' yields the
|
|||
|
|
rightmost code that passes _is_valid_code (DLI6), skipping the RME hop.
|
|||
|
|
"""
|
|||
|
|
for tok in reversed(_CODE_TOKEN_RE.findall(text)):
|
|||
|
|
if _is_valid_code(tok):
|
|||
|
|
return tok
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_name_leading(name):
|
|||
|
|
"""Leading shape (highest priority): the name OPENS with a standalone
|
|||
|
|
code-shaped token containing >=1 digit, delimited by space/dash/'(' --
|
|||
|
|
'WPT2 - Amazon...' -> WPT2, 'QDE1 (Co-located inside WDE1)' -> QDE1.
|
|||
|
|
|
|||
|
|
The digit requirement + leading position outrank the parens shape, so a
|
|||
|
|
'QDE1 (... WDE1)' resolves to the opener QDE1, not the host building WDE1.
|
|||
|
|
"""
|
|||
|
|
if not isinstance(name, str):
|
|||
|
|
return None
|
|||
|
|
m = _LEADING_CODE_RE.match(name.strip()[:_MAX_NAME])
|
|||
|
|
if not m:
|
|||
|
|
return None
|
|||
|
|
tok = m.group(1)
|
|||
|
|
if _is_valid_code(tok) and any(c.isdigit() for c in tok):
|
|||
|
|
return tok
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_name_parens(name):
|
|||
|
|
"""Shape 1: ship-to name in parentheses -- 'Services LLC (KLAL)' -> KLAL.
|
|||
|
|
|
|||
|
|
Uses the rightmost-non-skip resolver (same as ATTN) so parens content like
|
|||
|
|
'(ATTN: Wagon Wheel DS Station -WKY3)' resolves to WKY3, not the skip-listed
|
|||
|
|
ATTN hop.
|
|||
|
|
"""
|
|||
|
|
if not isinstance(name, str):
|
|||
|
|
return None
|
|||
|
|
for content in _PAREN_RE.findall(name[:_MAX_NAME]):
|
|||
|
|
code = _last_valid_code(content)
|
|||
|
|
if code:
|
|||
|
|
return code
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_name_after_dash(name):
|
|||
|
|
"""Shape 2: ship-to name with a dash.
|
|||
|
|
|
|||
|
|
Real ship-to names carry the code on EITHER side of the dash -- 'Amazon...
|
|||
|
|
LLC - SNY5' (after) and 'WPT2 - Amazon... LLC' (before). First prefer any
|
|||
|
|
segment that IS exactly a code-shaped token (leftmost such segment wins);
|
|||
|
|
only if no whole segment is an exact code, fall back to scanning the tail
|
|||
|
|
after the last dash (so a code-shaped word before the dash, e.g. 'LLC',
|
|||
|
|
can never leak in when the real after-dash code is skip-listed).
|
|||
|
|
"""
|
|||
|
|
if not isinstance(name, str):
|
|||
|
|
return None
|
|||
|
|
s = name[:_MAX_NAME]
|
|||
|
|
idx = max(s.rfind(d) for d in _DASHES)
|
|||
|
|
if idx == -1:
|
|||
|
|
return None
|
|||
|
|
for seg in _DASH_SPLIT_RE.split(s):
|
|||
|
|
tok = seg.strip()
|
|||
|
|
if _STRICT_CODE_RE.fullmatch(tok) and _is_valid_code(tok):
|
|||
|
|
return tok
|
|||
|
|
return _last_valid_code(s[idx + 1 :])
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_name_midtoken(name):
|
|||
|
|
"""Mid-name shape (lower priority): a standalone uppercase code-shaped token
|
|||
|
|
with >=1 digit anywhere in the name -- 'Amazon Fresh UVA5 Non-Inv (Prime)'
|
|||
|
|
-> UVA5. The digit requirement means all-letter words (FRESH, ...) can never
|
|||
|
|
match; mixed-case words never match the uppercase token shape at all.
|
|||
|
|
"""
|
|||
|
|
if not isinstance(name, str):
|
|||
|
|
return None
|
|||
|
|
for tok in _CODE_TOKEN_RE.findall(name[:_MAX_NAME]):
|
|||
|
|
if _is_valid_code(tok) and any(c.isdigit() for c in tok):
|
|||
|
|
return tok
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_from_attn(attn):
|
|||
|
|
"""Shapes 3 & 4: ship-to ATTN line (with dash/en-dash, or direct).
|
|||
|
|
|
|||
|
|
Unified: the rightmost non-skip code-shaped token across the whole ATTN
|
|||
|
|
value. Covers 'Wagon Wheel DS Station -WKY3' -> WKY3, 'HJX1' -> HJX1, and
|
|||
|
|
the multi-hop 'CBRE - RME - DLI6' -> DLI6.
|
|||
|
|
"""
|
|||
|
|
if not isinstance(attn, str):
|
|||
|
|
return None
|
|||
|
|
return _last_valid_code(attn[:_MAX_NAME])
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_name_is(name):
|
|||
|
|
"""Shape 5: the ship-to name IS the code -- 'DBU2' -> DBU2."""
|
|||
|
|
if not isinstance(name, str):
|
|||
|
|
return None
|
|||
|
|
n = name.strip()
|
|||
|
|
if _STRICT_CODE_RE.fullmatch(n) and _is_valid_code(n):
|
|||
|
|
return n
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_desc_prefix(desc):
|
|||
|
|
"""Shape 6: line-item description prefix -- 'DYO1 - ... - ...' -> DYO1."""
|
|||
|
|
if not isinstance(desc, str):
|
|||
|
|
return None
|
|||
|
|
m = _PREFIX_CODE_RE.match(desc[:80])
|
|||
|
|
if m and _is_valid_code(m.group(1)):
|
|||
|
|
return m.group(1)
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_desc_bracket(desc):
|
|||
|
|
"""Shape 7: line-item description in brackets -- '[HMK4] ...' -> HMK4."""
|
|||
|
|
if not isinstance(desc, str):
|
|||
|
|
return None
|
|||
|
|
for m in _BRACKET_CODE_RE.finditer(desc[:_MAX_DESC]):
|
|||
|
|
if _is_valid_code(m.group(1)):
|
|||
|
|
return m.group(1)
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _code_desc_freetoken(desc):
|
|||
|
|
"""Shape 8 (lowest priority): a standalone uppercase token, length 4-5,
|
|||
|
|
letter-first, with >=1 digit, anywhere in a description -- 'Need Sea Haven
|
|||
|
|
to pump out waste water aqt DSF7' -> DSF7. The length-4 floor (vs the 3-char
|
|||
|
|
floor of other shapes) and the digit requirement keep this loose free-scan
|
|||
|
|
from grabbing 3-letter uppercase words or all-letter tokens mid-sentence.
|
|||
|
|
"""
|
|||
|
|
if not isinstance(desc, str):
|
|||
|
|
return None
|
|||
|
|
for tok in _FREE_CODE_TOKEN_RE.findall(desc[:_MAX_DESC]):
|
|||
|
|
if _is_valid_code(tok) and any(c.isdigit() for c in tok):
|
|||
|
|
return tok
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _iter_items(parsed):
|
|||
|
|
"""Yield up to _MAX_ITEMS dict line items from parsed; tolerant of junk."""
|
|||
|
|
items = parsed.get("line_items")
|
|||
|
|
if not isinstance(items, list):
|
|||
|
|
return
|
|||
|
|
for item in items[:_MAX_ITEMS]:
|
|||
|
|
if isinstance(item, dict):
|
|||
|
|
yield item
|
|||
|
|
|
|||
|
|
|
|||
|
|
def derive_site_code(parsed):
|
|||
|
|
"""Derive the Amazon facility site_code, or None. NEVER raises.
|
|||
|
|
|
|||
|
|
Shapes are tried in strict priority order. A skip-listed sole candidate for
|
|||
|
|
a shape yields None for that shape and evaluation continues to the next:
|
|||
|
|
|
|||
|
|
leading -- name opens with a digit-bearing code ('WPT2 - ...')
|
|||
|
|
1 parens -- code inside '(...)'
|
|||
|
|
2 dash -- exact code-shaped segment either side of a dash
|
|||
|
|
3/4 attn -- rightmost non-skip code in the ATTN line
|
|||
|
|
5 name-is-- the name IS the code
|
|||
|
|
midtoken -- a digit-bearing code anywhere in the name
|
|||
|
|
6 prefix -- description leading code
|
|||
|
|
7 bracket-- description '[CODE]'
|
|||
|
|
8 free -- a length 4-5 digit-bearing code anywhere in a description
|
|||
|
|
"""
|
|||
|
|
try:
|
|||
|
|
if not isinstance(parsed, dict):
|
|||
|
|
return None
|
|||
|
|
ship_to = parsed.get("ship_to")
|
|||
|
|
if not isinstance(ship_to, dict):
|
|||
|
|
ship_to = {}
|
|||
|
|
name = ship_to.get("name")
|
|||
|
|
attn = ship_to.get("attn")
|
|||
|
|
|
|||
|
|
for shape in (
|
|||
|
|
_code_name_leading(name), # leading (highest)
|
|||
|
|
_code_name_parens(name), # 1
|
|||
|
|
_code_name_after_dash(name), # 2
|
|||
|
|
_code_from_attn(attn), # 3 & 4
|
|||
|
|
_code_name_is(name), # 5
|
|||
|
|
_code_name_midtoken(name), # mid-name
|
|||
|
|
):
|
|||
|
|
if shape:
|
|||
|
|
return shape
|
|||
|
|
|
|||
|
|
for item in _iter_items(parsed): # 6
|
|||
|
|
code = _code_desc_prefix(item.get("description"))
|
|||
|
|
if code:
|
|||
|
|
return code
|
|||
|
|
for item in _iter_items(parsed): # 7
|
|||
|
|
code = _code_desc_bracket(item.get("description"))
|
|||
|
|
if code:
|
|||
|
|
return code
|
|||
|
|
for item in _iter_items(parsed): # 8
|
|||
|
|
code = _code_desc_freetoken(item.get("description"))
|
|||
|
|
if code:
|
|||
|
|
return code
|
|||
|
|
return None
|
|||
|
|
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ---------------------------------------------------------------------------
|
|||
|
|
# fiscal_year
|
|||
|
|
# ---------------------------------------------------------------------------
|
|||
|
|
# A standalone 4-digit 20xx year. Word-bounded so it never matches digits buried
|
|||
|
|
# inside a longer run (e.g. the '2062' inside a PO id '18206023').
|
|||
|
|
_YEAR_RE = re.compile(r"\b(20\d{2})\b")
|
|||
|
|
# MM/DD/YY -> the trailing 2-digit year is expanded to 20YY.
|
|||
|
|
_SHORT_DATE_RE = re.compile(r"\b\d{1,2}/\d{1,2}/(\d{2})\b")
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _year_from_date(value):
|
|||
|
|
"""Extract a 4-digit year from a date string: a literal 20xx, else a
|
|||
|
|
MM/DD/YY whose YY expands to 20YY. Returns the 4-digit string or None."""
|
|||
|
|
if not isinstance(value, str):
|
|||
|
|
return None
|
|||
|
|
s = value[:_MAX_DATE]
|
|||
|
|
m = _YEAR_RE.search(s)
|
|||
|
|
if m:
|
|||
|
|
return m.group(1)
|
|||
|
|
m = _SHORT_DATE_RE.search(s)
|
|||
|
|
if m:
|
|||
|
|
return "20" + m.group(1)
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _year_from_text(value):
|
|||
|
|
"""A standalone 20xx year anywhere in free text, or None."""
|
|||
|
|
if not isinstance(value, str):
|
|||
|
|
return None
|
|||
|
|
m = _YEAR_RE.search(value[:_MAX_DESC])
|
|||
|
|
return m.group(1) if m else None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def derive_fiscal_year(parsed):
|
|||
|
|
"""Derive the 4-digit fiscal_year string, or None. NEVER raises.
|
|||
|
|
|
|||
|
|
Fallback order: (1) order_date year, (2) any line-item need_by year, (3) a
|
|||
|
|
standalone 20xx year in any line-item description.
|
|||
|
|
"""
|
|||
|
|
try:
|
|||
|
|
if not isinstance(parsed, dict):
|
|||
|
|
return None
|
|||
|
|
year = _year_from_date(parsed.get("order_date"))
|
|||
|
|
if year:
|
|||
|
|
return year
|
|||
|
|
for item in _iter_items(parsed):
|
|||
|
|
year = _year_from_date(item.get("need_by"))
|
|||
|
|
if year:
|
|||
|
|
return year
|
|||
|
|
for item in _iter_items(parsed):
|
|||
|
|
year = _year_from_text(item.get("description"))
|
|||
|
|
if year:
|
|||
|
|
return year
|
|||
|
|
return None
|
|||
|
|
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ---------------------------------------------------------------------------
|
|||
|
|
# trade
|
|||
|
|
# ---------------------------------------------------------------------------
|
|||
|
|
# Keyword matching interpretation (judgment call, documented): each keyword is
|
|||
|
|
# matched with a LEADING word boundary and no trailing boundary -- i.e. it hits
|
|||
|
|
# the keyword as a whole word OR as the prefix of a longer word. This makes
|
|||
|
|
# 'sign' match 'signage', 'dock door' match 'dock doors', and 'roof' match
|
|||
|
|
# 'roofing' (desired), while NOT matching a keyword buried mid-word ('ice' in
|
|||
|
|
# 'service'/'price', 'lock' in 'block'/'clock', 'gate' in 'mitigate') -- the
|
|||
|
|
# catastrophic false positives a bare substring test would produce. All matching
|
|||
|
|
# is done on a lower-cased, length-capped copy of the description.
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _kw(*words):
|
|||
|
|
"""Compile a linear leading-boundary alternation of literal keywords."""
|
|||
|
|
body = "|".join(re.escape(w) for w in words)
|
|||
|
|
return re.compile(r"\b(?:" + body + r")")
|
|||
|
|
|
|||
|
|
|
|||
|
|
_PM_RE = _kw(
|
|||
|
|
"plumbing pm",
|
|||
|
|
"plumbing preventative",
|
|||
|
|
"plumbing maintenance",
|
|||
|
|
"plumbing - backflow",
|
|||
|
|
"plumbing - water heater",
|
|||
|
|
)
|
|||
|
|
_PLUMBING_RE = _kw("plumbing")
|
|||
|
|
# BBM-structured 'Plumbing - <Fixture> - <Action> - BBM' row: 'plumbing' opening
|
|||
|
|
# the description immediately followed by a dash. Resolved by a measured ladder
|
|||
|
|
# (see _classify_desc) that matches the LLM baseline over the full corpus.
|
|||
|
|
_BBM_PLUMBING_RE = re.compile(r"plumbing\s*[-–—]")
|
|||
|
|
# Ladder rungs for a BBM plumbing row, checked in order:
|
|||
|
|
# (a) emergency/reactive -> Reactive
|
|||
|
|
_BBM_PLUMBING_REACTIVE_WORD_RE = _kw("reactive", "emergency")
|
|||
|
|
# (b) technician -> PM
|
|||
|
|
_BBM_PLUMBING_TECH_RE = _kw("technician")
|
|||
|
|
# (c) water heater / backflow / water fountain -> PM (before (d), so a
|
|||
|
|
# 'water heater repair' hits PM here and not Reactive on 'repair')
|
|||
|
|
_BBM_PLUMBING_PM_RE = _kw("water heater", "backflow", "water fountain")
|
|||
|
|
# (d) clog/unclog/leak/repair/sewer/drain -> Reactive ('clog'/'leak' prefixes
|
|||
|
|
# also catch 'clogged'/'leaking')
|
|||
|
|
_BBM_PLUMBING_REACTIVE_RE = _kw("clog", "unclog", "leak", "repair", "sewer", "drain")
|
|||
|
|
# (e) else (project, install, misc) -> PM
|
|||
|
|
# Plumbing-specific qualifiers -- these indicate Plumbing - Reactive on their
|
|||
|
|
# OWN, without the word 'plumbing' present ('CLOGGED PIT AUGER' -> Reactive).
|
|||
|
|
_PLUMBING_SPECIFIC_RE = _kw(
|
|||
|
|
"clog",
|
|||
|
|
"unclog",
|
|||
|
|
"sewer",
|
|||
|
|
"drain",
|
|||
|
|
"grease trap",
|
|||
|
|
"jetter",
|
|||
|
|
"toilet",
|
|||
|
|
"faucet",
|
|||
|
|
"urinal",
|
|||
|
|
)
|
|||
|
|
# Generic reactive qualifiers -- these require the word 'plumbing' to be present
|
|||
|
|
# before they route to Plumbing - Reactive.
|
|||
|
|
_PLUMBING_GENERIC_RE = _kw(
|
|||
|
|
"reactive",
|
|||
|
|
"emergency",
|
|||
|
|
"repair",
|
|||
|
|
"leak",
|
|||
|
|
"flood",
|
|||
|
|
"water line",
|
|||
|
|
"pipe",
|
|||
|
|
)
|
|||
|
|
_ELECTRICAL_RE = _kw(
|
|||
|
|
"electrical",
|
|||
|
|
"lighting",
|
|||
|
|
"ballast",
|
|||
|
|
"outlet",
|
|||
|
|
"circuit",
|
|||
|
|
"panel",
|
|||
|
|
"generator",
|
|||
|
|
"transformer",
|
|||
|
|
"conduit",
|
|||
|
|
)
|
|||
|
|
_HVAC_RE = _kw(
|
|||
|
|
"hvac",
|
|||
|
|
"heating",
|
|||
|
|
"cooling",
|
|||
|
|
"air conditioning",
|
|||
|
|
"rtu",
|
|||
|
|
"ahu",
|
|||
|
|
"vav",
|
|||
|
|
"refrigerant",
|
|||
|
|
"thermostat",
|
|||
|
|
"ductwork",
|
|||
|
|
)
|
|||
|
|
_DOCK_DOORS_RE = _kw(
|
|||
|
|
"dock door",
|
|||
|
|
"dock leveler",
|
|||
|
|
"dock plate",
|
|||
|
|
"dock seal",
|
|||
|
|
"dock bumper",
|
|||
|
|
)
|
|||
|
|
_DOORS_RE = _kw("door", "overhead door", "roll-up", "automatic door", "access door")
|
|||
|
|
_SIGNAGE_RE = _kw("sign", "banner", "wayfinding", "marquee", "directional")
|
|||
|
|
_CARPENTRY_RE = _kw("carpentry", "cabinet", "millwork", "trim", "shelving", "framing")
|
|||
|
|
_FENCE_RE = _kw("fence", "fencing", "bollard")
|
|||
|
|
_GATE_RE = _kw("gate")
|
|||
|
|
_CONVEYANCE_RE = _kw("conveyor", "conveyance", "mhe", "material handling", "sortation")
|
|||
|
|
_PAINTING_RE = _kw("paint", "painting", "primer", "coating", "touch-up")
|
|||
|
|
_FLOORING_RE = _kw("floor", "tile", "carpet", "epoxy", "polishing")
|
|||
|
|
_JANITORIAL_RE = _kw(
|
|||
|
|
"janitorial", "cleaning", "custodial", "pressure wash", "power wash"
|
|||
|
|
)
|
|||
|
|
_FIRE_RE = _kw("fire", "sprinkler", "extinguisher", "fire alarm", "suppression")
|
|||
|
|
_LANDSCAPING_RE = _kw("landscape", "lawn", "tree", "yard", "mowing", "irrigation")
|
|||
|
|
_ROOFING_RE = _kw("roof", "roofing", "gutter", "downspout")
|
|||
|
|
# Security/Locksmith: 'lock' and 'key' match as EXACT whole words only (so
|
|||
|
|
# 'Locker' is not 'lock' and 'keyboard' is not 'key'); the multi-word terms keep
|
|||
|
|
# the leading-boundary/prefix behavior. The phrase 'key box' is stripped before
|
|||
|
|
# this test (see _classify_desc) so a Key Box fixture never triggers Locksmith.
|
|||
|
|
_SECURITY_RE = re.compile(r"\b(?:lock|key)\b|\b(?:access control|camera|security|cctv)")
|
|||
|
|
_SNOW_RE = _kw("snow", "ice", "salt", "de-ice", "plow")
|
|||
|
|
_EMERGENCY_RE = _kw("emergency")
|
|||
|
|
# General Building catch-all sub-shapes.
|
|||
|
|
_HANDYMAN_RE = re.compile(r"\bhandyman\b")
|
|||
|
|
_PROJECT_RE = re.compile(r"\bproject\b")
|
|||
|
|
|
|||
|
|
TRADE_UPLIFT = "PO Uplift"
|
|||
|
|
TRADE_GENERAL_BUILDING = "General Building"
|
|||
|
|
|
|||
|
|
# Simple (single-regex) trades, in strict priority order. The compound and
|
|||
|
|
# exception-bearing trades (Plumbing, Electrical, Fencing/Gates, the General
|
|||
|
|
# Building family, PO Uplift) are handled inline in _classify_desc.
|
|||
|
|
_SIMPLE_TRADES = (
|
|||
|
|
(_HVAC_RE, "HVAC"),
|
|||
|
|
(_DOCK_DOORS_RE, "Dock Doors"),
|
|||
|
|
(_DOORS_RE, "Doors"),
|
|||
|
|
(_SIGNAGE_RE, "Signage"),
|
|||
|
|
(_CARPENTRY_RE, "Carpentry"),
|
|||
|
|
)
|
|||
|
|
# Simple trades that follow Fencing/Gates in the priority table. Security and
|
|||
|
|
# Snow are handled explicitly after this loop (Security needs the 'key box'
|
|||
|
|
# exclusion applied to its match target, so it can't share the generic loop).
|
|||
|
|
_SIMPLE_TRADES_TAIL = (
|
|||
|
|
(_CONVEYANCE_RE, "Conveyance/MHE"),
|
|||
|
|
(_PAINTING_RE, "Painting"),
|
|||
|
|
(_FLOORING_RE, "Flooring"),
|
|||
|
|
(_JANITORIAL_RE, "Janitorial"),
|
|||
|
|
(_FIRE_RE, "Fire/Life Safety"),
|
|||
|
|
(_LANDSCAPING_RE, "Landscaping/Yard"),
|
|||
|
|
(_ROOFING_RE, "Roofing"),
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _classify_desc(desc):
|
|||
|
|
"""Classify a single non-empty description into one trade label.
|
|||
|
|
|
|||
|
|
Returns a label for any non-empty description -- 'General Building' is the
|
|||
|
|
catch-all when nothing else matches. Callers must not pass empty/blank
|
|||
|
|
descriptions (that case is handled upstream so it can return None).
|
|||
|
|
"""
|
|||
|
|
low = desc[:_MAX_TRADE].lower()
|
|||
|
|
stripped = low.strip()
|
|||
|
|
|
|||
|
|
# PO Uplift: exactly or primarily "PO Uplift".
|
|||
|
|
if stripped == "po uplift" or stripped.startswith("po uplift"):
|
|||
|
|
return TRADE_UPLIFT
|
|||
|
|
|
|||
|
|
# BBM-structured 'Plumbing - ...' row: resolved by a measured ladder tuned to
|
|||
|
|
# the LLM baseline over the full corpus (first rung wins).
|
|||
|
|
if _BBM_PLUMBING_RE.match(stripped):
|
|||
|
|
if _BBM_PLUMBING_REACTIVE_WORD_RE.search(low): # (a) emergency/reactive
|
|||
|
|
return "Plumbing - Reactive"
|
|||
|
|
if _BBM_PLUMBING_TECH_RE.search(low): # (b) technician
|
|||
|
|
return "Plumbing - PM"
|
|||
|
|
if _BBM_PLUMBING_PM_RE.search(low): # (c) heater/backflow/fountain
|
|||
|
|
return "Plumbing - PM"
|
|||
|
|
if _BBM_PLUMBING_REACTIVE_RE.search(low): # (d) clog/leak/repair/...
|
|||
|
|
return "Plumbing - Reactive"
|
|||
|
|
return "Plumbing - PM" # (e) project/install/misc
|
|||
|
|
# Plumbing - PM (before Reactive) for non-BBM free text.
|
|||
|
|
if _PM_RE.search(low):
|
|||
|
|
return "Plumbing - PM"
|
|||
|
|
# Plumbing - Reactive: a plumbing-specific qualifier ALONE (clog, drain,
|
|||
|
|
# toilet, ...), or the word 'plumbing' PLUS a generic reactive qualifier.
|
|||
|
|
if _PLUMBING_SPECIFIC_RE.search(low):
|
|||
|
|
return "Plumbing - Reactive"
|
|||
|
|
if _PLUMBING_RE.search(low) and _PLUMBING_GENERIC_RE.search(low):
|
|||
|
|
return "Plumbing - Reactive"
|
|||
|
|
|
|||
|
|
# Electrical -- but electrical keywords do NOT match in a dock-door context.
|
|||
|
|
if _ELECTRICAL_RE.search(low) and "dock door" not in low:
|
|||
|
|
return "Electrical"
|
|||
|
|
|
|||
|
|
for regex, label in _SIMPLE_TRADES:
|
|||
|
|
if regex.search(low):
|
|||
|
|
return label
|
|||
|
|
|
|||
|
|
# Fencing/Gates: fence/fencing/bollard anywhere, or 'gate' but NOT 'dock
|
|||
|
|
# gate' (the dock-gate occurrences are removed before the gate test).
|
|||
|
|
if _FENCE_RE.search(low) or _GATE_RE.search(low.replace("dock gate", " ")):
|
|||
|
|
return "Fencing/Gates"
|
|||
|
|
|
|||
|
|
for regex, label in _SIMPLE_TRADES_TAIL:
|
|||
|
|
if regex.search(low):
|
|||
|
|
return label
|
|||
|
|
|
|||
|
|
# Security/Locksmith: 'lock'/'key' as whole words, with 'key box' stripped
|
|||
|
|
# first (mirrors the 'dock gate' exclusion) so a Key Box fixture is not one.
|
|||
|
|
if _SECURITY_RE.search(low.replace("key box", " ")):
|
|||
|
|
return "Security/Locksmith"
|
|||
|
|
if _SNOW_RE.search(low):
|
|||
|
|
return "Snow Removal"
|
|||
|
|
|
|||
|
|
# General Building family (catch-alls, in priority order).
|
|||
|
|
if stripped.startswith("emer") or _EMERGENCY_RE.search(low):
|
|||
|
|
return "General Building - Emergency"
|
|||
|
|
if _HANDYMAN_RE.search(low) or "general building technician" in low:
|
|||
|
|
return "General Building - Handyman"
|
|||
|
|
if "general building project" in low or _PROJECT_RE.search(low):
|
|||
|
|
return "General Building - Project"
|
|||
|
|
# Keyword-less BBM 'General Building - <Fixture> - ...' rows (parking lot,
|
|||
|
|
# fan, ceiling, television, locker, key box, ...) fall through here to the
|
|||
|
|
# plain catch-all -- that matches the measured LLM majority for such rows.
|
|||
|
|
return TRADE_GENERAL_BUILDING
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _to_amount(value):
|
|||
|
|
"""Coerce a line amount (Decimal | str | int/float | junk) to a Decimal for
|
|||
|
|
comparison. Anything unparseable becomes Decimal(0) -- defensive, never
|
|||
|
|
raises, so an item with a bad amount simply cannot win the highest-value
|
|||
|
|
tie-break."""
|
|||
|
|
if isinstance(value, bool):
|
|||
|
|
return Decimal(0)
|
|||
|
|
if isinstance(value, Decimal):
|
|||
|
|
return value if value.is_finite() else Decimal(0)
|
|||
|
|
if isinstance(value, int):
|
|||
|
|
return Decimal(value)
|
|||
|
|
if isinstance(value, float):
|
|||
|
|
try:
|
|||
|
|
d = Decimal(str(value))
|
|||
|
|
return d if d.is_finite() else Decimal(0)
|
|||
|
|
except InvalidOperation:
|
|||
|
|
return Decimal(0)
|
|||
|
|
if isinstance(value, str):
|
|||
|
|
try:
|
|||
|
|
d = Decimal(value.replace(",", "").strip())
|
|||
|
|
return d if d.is_finite() else Decimal(0)
|
|||
|
|
except (InvalidOperation, ValueError):
|
|||
|
|
return Decimal(0)
|
|||
|
|
return Decimal(0)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def derive_trade(parsed):
|
|||
|
|
"""Derive the PO's primary trade label, or None. NEVER raises.
|
|||
|
|
|
|||
|
|
Each line item with a usable description is classified. The PO trade is the
|
|||
|
|
primary non-"PO Uplift" trade; when several distinct non-uplift trades are
|
|||
|
|
present, the trade of the highest-`amount` non-uplift item wins. If every
|
|||
|
|
described item is PO Uplift, returns "PO Uplift". With no line items or no
|
|||
|
|
usable descriptions at all, returns None ("General Building" is only ever
|
|||
|
|
returned for a description that matched nothing).
|
|||
|
|
"""
|
|||
|
|
try:
|
|||
|
|
if not isinstance(parsed, dict):
|
|||
|
|
return None
|
|||
|
|
classified = [] # (trade_label, amount_decimal)
|
|||
|
|
# Aggregate CPU budget across ALL items: the per-item _MAX_TRADE cap
|
|||
|
|
# alone still allows _MAX_ITEMS x _MAX_TRADE x ~25 regex passes
|
|||
|
|
# (~10s full-core, minutes under the 256MB Lambda's CPU throttle ->
|
|||
|
|
# timeout -> async retry -> DLQ). Real POs total well under 10k chars
|
|||
|
|
# of description text; stop classifying once the budget is spent and
|
|||
|
|
# decide from what was classified (sh-security-review
|
|||
|
|
# PO-DERIVED-AVAIL-001 defense-in-depth).
|
|||
|
|
budget = _MAX_TRADE_TOTAL
|
|||
|
|
for item in _iter_items(parsed):
|
|||
|
|
desc = item.get("description")
|
|||
|
|
if not isinstance(desc, str) or not desc.strip():
|
|||
|
|
continue
|
|||
|
|
if budget <= 0:
|
|||
|
|
break
|
|||
|
|
desc = desc[:budget]
|
|||
|
|
budget -= len(desc)
|
|||
|
|
classified.append((_classify_desc(desc), _to_amount(item.get("amount"))))
|
|||
|
|
if not classified:
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
non_uplift = [(t, a) for (t, a) in classified if t != TRADE_UPLIFT]
|
|||
|
|
if not non_uplift:
|
|||
|
|
return TRADE_UPLIFT
|
|||
|
|
if len({t for (t, _) in non_uplift}) == 1:
|
|||
|
|
return non_uplift[0][0]
|
|||
|
|
best_trade, best_amount = non_uplift[0]
|
|||
|
|
for trade, amount in non_uplift[1:]:
|
|||
|
|
if amount > best_amount:
|
|||
|
|
best_trade, best_amount = trade, amount
|
|||
|
|
return best_trade
|
|||
|
|
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def derive_all(parsed):
|
|||
|
|
"""All three derived fields as a dict. NEVER raises: on any failure returns
|
|||
|
|
an all-None dict so the caller always gets the three keys."""
|
|||
|
|
try:
|
|||
|
|
return {
|
|||
|
|
"site_code": derive_site_code(parsed),
|
|||
|
|
"trade": derive_trade(parsed),
|
|||
|
|
"fiscal_year": derive_fiscal_year(parsed),
|
|||
|
|
}
|
|||
|
|
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
|||
|
|
return {"site_code": None, "trade": None, "fiscal_year": None}
|