mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-09-30 13:03:14 +00:00
Some checks are pending
Deploy / deploy (push) Waiting to run
* feat: Python derived-field classifier with shadow telemetry for PO ingest Port the site_code/trade/fiscal_year rules from EXTRACTION_PROMPT into a pure, total derived_fields module applied in the shared enrich_parsed() post-stage. Python fills gaps on both parse paths (the template path has no LLM values, closing the derived-field gap opened by the PR #105 two-PR split) and never overwrites a non-null LLM value; on ai_fallback a DerivedFieldAgreement EMF record per field shadows Python against the LLM during the bake. Rules hardened against a full-corpus backtest (3,422 real emails vs the LLM-written baseline): site_code 99.4% with zero Python-wrong cases, fiscal_year 100%, trade 96.8% ex-deliberate. Also: quantity/price now declared numeric in the prompt, and derived_fields.py added to the po_stack bundling copy (deploy-time ImportError otherwise). * fix: security-review hardening — EMF value length clamp, aggregate trade CPU budget sh-security-review (4 detectors + proof-or-kill verifier): PASS, 0 confirmed critical/high. Fixes the one confirmed low (unbounded LLM-value str() into the DerivedFieldAgreement EMF log line, clamped to 64 chars) and adds the verifier-recommended defense-in-depth aggregate character budget across line items in derive_trade (per-item caps alone allowed ~10s full-core on a pathological direct-call input; unreachable through the deployed handler but cheap to bound). Bundling cp list now carries a warning comment (GPT-4.1 cross-review FIX).
676 lines
26 KiB
Python
676 lines
26 KiB
Python
"""Deterministic derived-field classifiers for Coupa purchase order emails.
|
||
|
||
Pure module: stdlib only -- no boto3, no network, no imports from handler or
|
||
template_parser. Computes the three DERIVED_KEYS (site_code, fiscal_year, trade)
|
||
that the template parser and the LLM path both leave for a shared post-stage to
|
||
fill identically (see template_parser.py DERIVED_KEYS / enrich_parsed).
|
||
|
||
This is a FAITHFUL v1 port of the English rules in the handler's
|
||
EXTRACTION_PROMPT sections "## site_code extraction", "## fiscal_year" and
|
||
"## trade classification". A corpus backtest will harden the thresholds later;
|
||
this module intentionally invents no rule beyond those three sections.
|
||
|
||
TOTALITY (hard requirement): this runs on the untrusted email path in an
|
||
S3-async Lambda, where an uncaught exception means infinite retry -> DLQ ->
|
||
silent data loss. Every public function therefore mirrors
|
||
template_parser.try_deterministic_parse: it wraps its whole body in a broad
|
||
try/except and NEVER raises -- a bad/hostile input yields None (or an all-None
|
||
dict for derive_all), never a traceback. Inputs are treated as hostile: wrong
|
||
types, None, non-dict line items and multi-megabyte strings are all tolerated.
|
||
Scans are length-capped and every regex is linear (no nested quantifiers, no
|
||
catastrophic backtracking).
|
||
|
||
Public API:
|
||
derive_site_code(parsed) -> str | None
|
||
derive_fiscal_year(parsed) -> str | None
|
||
derive_trade(parsed) -> str | None
|
||
derive_all(parsed) -> {"site_code":..., "trade":..., "fiscal_year":...}
|
||
|
||
`parsed` is the post-extraction contract dict from either path. Relevant inputs:
|
||
parsed["ship_to"]["name"], parsed["ship_to"]["attn"],
|
||
parsed["line_items"][*]["description"|"amount"|"need_by"], parsed["order_date"].
|
||
"""
|
||
|
||
import re
|
||
from decimal import Decimal, InvalidOperation
|
||
|
||
# --- scan caps (hostile-input blast-radius limits) ---------------------------
|
||
_MAX_ITEMS = 500 # line_items scanned at most
|
||
_MAX_NAME = 4096 # ship_to name/attn chars scanned for a site code
|
||
_MAX_DESC = 4096 # description chars scanned for a bracketed/standalone code
|
||
_MAX_DATE = 1024 # date-string chars scanned for a year
|
||
_MAX_TRADE = 100_000 # description chars scanned for trade keywords (per item)
|
||
_MAX_TRADE_TOTAL = 200_000 # aggregate description chars classified per PO
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# site_code
|
||
# ---------------------------------------------------------------------------
|
||
# Token shape: 3-5 chars, uppercase A-Z0-9, MUST start with a letter. Codes may
|
||
# be all letters (e.g. KLAL) -- a digit is NOT required.
|
||
_STRICT_CODE_RE = re.compile(r"[A-Z][A-Z0-9]{2,4}")
|
||
# Same shape, but only as a standalone token (not embedded in a longer
|
||
# alphanumeric run): '–WKY3' matches, 'Station' does not, 'DLI6X99' does not.
|
||
_CODE_TOKEN_RE = re.compile(r"(?<![A-Za-z0-9])[A-Z][A-Z0-9]{2,4}(?![A-Za-z0-9])")
|
||
_PAREN_RE = re.compile(r"\(([^()\n]{1,40})\)")
|
||
_BRACKET_CODE_RE = re.compile(r"\[([A-Z][A-Z0-9]{2,4})\]")
|
||
# Description prefix: leading code then a dash/en-dash/em-dash separator.
|
||
_PREFIX_CODE_RE = re.compile(r"\s*([A-Z][A-Z0-9]{2,4})\s*[-–—]")
|
||
# Leading code-shaped token that opens the ship-to name and is delimited by a
|
||
# space, dash or '(' -- e.g. 'WPT2 - Amazon...' / 'QDE1 (Co-located ...)'.
|
||
_LEADING_CODE_RE = re.compile(r"([A-Z][A-Z0-9]{2,4})(?=[\s\-–—(])")
|
||
# Free-token code in a description: standalone, length 4-5, letter-first.
|
||
_FREE_CODE_TOKEN_RE = re.compile(r"(?<![A-Za-z0-9])[A-Z][A-Z0-9]{3,4}(?![A-Za-z0-9])")
|
||
_DASHES = "-–—"
|
||
_DASH_SPLIT_RE = re.compile(r"[-–—]")
|
||
|
||
# Never a site_code (from the prompt's "Not site codes" list). If the only
|
||
# candidate for a shape is skip-listed, that shape yields None and evaluation
|
||
# falls through to the next lower-priority shape.
|
||
_SKIP = frozenset(
|
||
{
|
||
"RME",
|
||
"BBM",
|
||
"JLL",
|
||
"PARAG",
|
||
"ERIK",
|
||
"HVAC",
|
||
"LED",
|
||
"PVC",
|
||
"ADA",
|
||
"OSHA",
|
||
"EMR",
|
||
"BMS",
|
||
"DDC",
|
||
"MRO",
|
||
"NTE",
|
||
"EST",
|
||
# Corporate-suffix / addressing tokens -- never a site code.
|
||
"LLC",
|
||
"INC",
|
||
"CORP",
|
||
"LTD",
|
||
"ATTN",
|
||
}
|
||
)
|
||
|
||
|
||
def _is_valid_code(tok):
|
||
"""A code-shaped token is a real site code only if it is not skip-listed AND
|
||
carries >=2 alphabetic characters.
|
||
|
||
Measured over all 1,063 distinct LLM-extracted codes: 999 have 3 letters, 7
|
||
have 4, 3 have 5, and NONE have fewer than 2. Sub-2-letter tokens (e.g.
|
||
'B187') are PO-prefix-style garbage, so every site-code shape rejects them.
|
||
"""
|
||
return tok not in _SKIP and sum(c.isalpha() for c in tok) >= 2
|
||
|
||
|
||
def _last_valid_code(text):
|
||
"""Rightmost valid code-shaped token in `text`.
|
||
|
||
This is the multi-hop resolver: an ATTN like 'CBRE - RME - DLI6' yields the
|
||
rightmost code that passes _is_valid_code (DLI6), skipping the RME hop.
|
||
"""
|
||
for tok in reversed(_CODE_TOKEN_RE.findall(text)):
|
||
if _is_valid_code(tok):
|
||
return tok
|
||
return None
|
||
|
||
|
||
def _code_name_leading(name):
|
||
"""Leading shape (highest priority): the name OPENS with a standalone
|
||
code-shaped token containing >=1 digit, delimited by space/dash/'(' --
|
||
'WPT2 - Amazon...' -> WPT2, 'QDE1 (Co-located inside WDE1)' -> QDE1.
|
||
|
||
The digit requirement + leading position outrank the parens shape, so a
|
||
'QDE1 (... WDE1)' resolves to the opener QDE1, not the host building WDE1.
|
||
"""
|
||
if not isinstance(name, str):
|
||
return None
|
||
m = _LEADING_CODE_RE.match(name.strip()[:_MAX_NAME])
|
||
if not m:
|
||
return None
|
||
tok = m.group(1)
|
||
if _is_valid_code(tok) and any(c.isdigit() for c in tok):
|
||
return tok
|
||
return None
|
||
|
||
|
||
def _code_name_parens(name):
|
||
"""Shape 1: ship-to name in parentheses -- 'Services LLC (KLAL)' -> KLAL.
|
||
|
||
Uses the rightmost-non-skip resolver (same as ATTN) so parens content like
|
||
'(ATTN: Wagon Wheel DS Station -WKY3)' resolves to WKY3, not the skip-listed
|
||
ATTN hop.
|
||
"""
|
||
if not isinstance(name, str):
|
||
return None
|
||
for content in _PAREN_RE.findall(name[:_MAX_NAME]):
|
||
code = _last_valid_code(content)
|
||
if code:
|
||
return code
|
||
return None
|
||
|
||
|
||
def _code_name_after_dash(name):
|
||
"""Shape 2: ship-to name with a dash.
|
||
|
||
Real ship-to names carry the code on EITHER side of the dash -- 'Amazon...
|
||
LLC - SNY5' (after) and 'WPT2 - Amazon... LLC' (before). First prefer any
|
||
segment that IS exactly a code-shaped token (leftmost such segment wins);
|
||
only if no whole segment is an exact code, fall back to scanning the tail
|
||
after the last dash (so a code-shaped word before the dash, e.g. 'LLC',
|
||
can never leak in when the real after-dash code is skip-listed).
|
||
"""
|
||
if not isinstance(name, str):
|
||
return None
|
||
s = name[:_MAX_NAME]
|
||
idx = max(s.rfind(d) for d in _DASHES)
|
||
if idx == -1:
|
||
return None
|
||
for seg in _DASH_SPLIT_RE.split(s):
|
||
tok = seg.strip()
|
||
if _STRICT_CODE_RE.fullmatch(tok) and _is_valid_code(tok):
|
||
return tok
|
||
return _last_valid_code(s[idx + 1 :])
|
||
|
||
|
||
def _code_name_midtoken(name):
|
||
"""Mid-name shape (lower priority): a standalone uppercase code-shaped token
|
||
with >=1 digit anywhere in the name -- 'Amazon Fresh UVA5 Non-Inv (Prime)'
|
||
-> UVA5. The digit requirement means all-letter words (FRESH, ...) can never
|
||
match; mixed-case words never match the uppercase token shape at all.
|
||
"""
|
||
if not isinstance(name, str):
|
||
return None
|
||
for tok in _CODE_TOKEN_RE.findall(name[:_MAX_NAME]):
|
||
if _is_valid_code(tok) and any(c.isdigit() for c in tok):
|
||
return tok
|
||
return None
|
||
|
||
|
||
def _code_from_attn(attn):
|
||
"""Shapes 3 & 4: ship-to ATTN line (with dash/en-dash, or direct).
|
||
|
||
Unified: the rightmost non-skip code-shaped token across the whole ATTN
|
||
value. Covers 'Wagon Wheel DS Station -WKY3' -> WKY3, 'HJX1' -> HJX1, and
|
||
the multi-hop 'CBRE - RME - DLI6' -> DLI6.
|
||
"""
|
||
if not isinstance(attn, str):
|
||
return None
|
||
return _last_valid_code(attn[:_MAX_NAME])
|
||
|
||
|
||
def _code_name_is(name):
|
||
"""Shape 5: the ship-to name IS the code -- 'DBU2' -> DBU2."""
|
||
if not isinstance(name, str):
|
||
return None
|
||
n = name.strip()
|
||
if _STRICT_CODE_RE.fullmatch(n) and _is_valid_code(n):
|
||
return n
|
||
return None
|
||
|
||
|
||
def _code_desc_prefix(desc):
|
||
"""Shape 6: line-item description prefix -- 'DYO1 - ... - ...' -> DYO1."""
|
||
if not isinstance(desc, str):
|
||
return None
|
||
m = _PREFIX_CODE_RE.match(desc[:80])
|
||
if m and _is_valid_code(m.group(1)):
|
||
return m.group(1)
|
||
return None
|
||
|
||
|
||
def _code_desc_bracket(desc):
|
||
"""Shape 7: line-item description in brackets -- '[HMK4] ...' -> HMK4."""
|
||
if not isinstance(desc, str):
|
||
return None
|
||
for m in _BRACKET_CODE_RE.finditer(desc[:_MAX_DESC]):
|
||
if _is_valid_code(m.group(1)):
|
||
return m.group(1)
|
||
return None
|
||
|
||
|
||
def _code_desc_freetoken(desc):
|
||
"""Shape 8 (lowest priority): a standalone uppercase token, length 4-5,
|
||
letter-first, with >=1 digit, anywhere in a description -- 'Need Sea Haven
|
||
to pump out waste water aqt DSF7' -> DSF7. The length-4 floor (vs the 3-char
|
||
floor of other shapes) and the digit requirement keep this loose free-scan
|
||
from grabbing 3-letter uppercase words or all-letter tokens mid-sentence.
|
||
"""
|
||
if not isinstance(desc, str):
|
||
return None
|
||
for tok in _FREE_CODE_TOKEN_RE.findall(desc[:_MAX_DESC]):
|
||
if _is_valid_code(tok) and any(c.isdigit() for c in tok):
|
||
return tok
|
||
return None
|
||
|
||
|
||
def _iter_items(parsed):
|
||
"""Yield up to _MAX_ITEMS dict line items from parsed; tolerant of junk."""
|
||
items = parsed.get("line_items")
|
||
if not isinstance(items, list):
|
||
return
|
||
for item in items[:_MAX_ITEMS]:
|
||
if isinstance(item, dict):
|
||
yield item
|
||
|
||
|
||
def derive_site_code(parsed):
|
||
"""Derive the Amazon facility site_code, or None. NEVER raises.
|
||
|
||
Shapes are tried in strict priority order. A skip-listed sole candidate for
|
||
a shape yields None for that shape and evaluation continues to the next:
|
||
|
||
leading -- name opens with a digit-bearing code ('WPT2 - ...')
|
||
1 parens -- code inside '(...)'
|
||
2 dash -- exact code-shaped segment either side of a dash
|
||
3/4 attn -- rightmost non-skip code in the ATTN line
|
||
5 name-is-- the name IS the code
|
||
midtoken -- a digit-bearing code anywhere in the name
|
||
6 prefix -- description leading code
|
||
7 bracket-- description '[CODE]'
|
||
8 free -- a length 4-5 digit-bearing code anywhere in a description
|
||
"""
|
||
try:
|
||
if not isinstance(parsed, dict):
|
||
return None
|
||
ship_to = parsed.get("ship_to")
|
||
if not isinstance(ship_to, dict):
|
||
ship_to = {}
|
||
name = ship_to.get("name")
|
||
attn = ship_to.get("attn")
|
||
|
||
for shape in (
|
||
_code_name_leading(name), # leading (highest)
|
||
_code_name_parens(name), # 1
|
||
_code_name_after_dash(name), # 2
|
||
_code_from_attn(attn), # 3 & 4
|
||
_code_name_is(name), # 5
|
||
_code_name_midtoken(name), # mid-name
|
||
):
|
||
if shape:
|
||
return shape
|
||
|
||
for item in _iter_items(parsed): # 6
|
||
code = _code_desc_prefix(item.get("description"))
|
||
if code:
|
||
return code
|
||
for item in _iter_items(parsed): # 7
|
||
code = _code_desc_bracket(item.get("description"))
|
||
if code:
|
||
return code
|
||
for item in _iter_items(parsed): # 8
|
||
code = _code_desc_freetoken(item.get("description"))
|
||
if code:
|
||
return code
|
||
return None
|
||
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
||
return None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# fiscal_year
|
||
# ---------------------------------------------------------------------------
|
||
# A standalone 4-digit 20xx year. Word-bounded so it never matches digits buried
|
||
# inside a longer run (e.g. the '2062' inside a PO id '18206023').
|
||
_YEAR_RE = re.compile(r"\b(20\d{2})\b")
|
||
# MM/DD/YY -> the trailing 2-digit year is expanded to 20YY.
|
||
_SHORT_DATE_RE = re.compile(r"\b\d{1,2}/\d{1,2}/(\d{2})\b")
|
||
|
||
|
||
def _year_from_date(value):
|
||
"""Extract a 4-digit year from a date string: a literal 20xx, else a
|
||
MM/DD/YY whose YY expands to 20YY. Returns the 4-digit string or None."""
|
||
if not isinstance(value, str):
|
||
return None
|
||
s = value[:_MAX_DATE]
|
||
m = _YEAR_RE.search(s)
|
||
if m:
|
||
return m.group(1)
|
||
m = _SHORT_DATE_RE.search(s)
|
||
if m:
|
||
return "20" + m.group(1)
|
||
return None
|
||
|
||
|
||
def _year_from_text(value):
|
||
"""A standalone 20xx year anywhere in free text, or None."""
|
||
if not isinstance(value, str):
|
||
return None
|
||
m = _YEAR_RE.search(value[:_MAX_DESC])
|
||
return m.group(1) if m else None
|
||
|
||
|
||
def derive_fiscal_year(parsed):
|
||
"""Derive the 4-digit fiscal_year string, or None. NEVER raises.
|
||
|
||
Fallback order: (1) order_date year, (2) any line-item need_by year, (3) a
|
||
standalone 20xx year in any line-item description.
|
||
"""
|
||
try:
|
||
if not isinstance(parsed, dict):
|
||
return None
|
||
year = _year_from_date(parsed.get("order_date"))
|
||
if year:
|
||
return year
|
||
for item in _iter_items(parsed):
|
||
year = _year_from_date(item.get("need_by"))
|
||
if year:
|
||
return year
|
||
for item in _iter_items(parsed):
|
||
year = _year_from_text(item.get("description"))
|
||
if year:
|
||
return year
|
||
return None
|
||
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
||
return None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# trade
|
||
# ---------------------------------------------------------------------------
|
||
# Keyword matching interpretation (judgment call, documented): each keyword is
|
||
# matched with a LEADING word boundary and no trailing boundary -- i.e. it hits
|
||
# the keyword as a whole word OR as the prefix of a longer word. This makes
|
||
# 'sign' match 'signage', 'dock door' match 'dock doors', and 'roof' match
|
||
# 'roofing' (desired), while NOT matching a keyword buried mid-word ('ice' in
|
||
# 'service'/'price', 'lock' in 'block'/'clock', 'gate' in 'mitigate') -- the
|
||
# catastrophic false positives a bare substring test would produce. All matching
|
||
# is done on a lower-cased, length-capped copy of the description.
|
||
|
||
|
||
def _kw(*words):
|
||
"""Compile a linear leading-boundary alternation of literal keywords."""
|
||
body = "|".join(re.escape(w) for w in words)
|
||
return re.compile(r"\b(?:" + body + r")")
|
||
|
||
|
||
_PM_RE = _kw(
|
||
"plumbing pm",
|
||
"plumbing preventative",
|
||
"plumbing maintenance",
|
||
"plumbing - backflow",
|
||
"plumbing - water heater",
|
||
)
|
||
_PLUMBING_RE = _kw("plumbing")
|
||
# BBM-structured 'Plumbing - <Fixture> - <Action> - BBM' row: 'plumbing' opening
|
||
# the description immediately followed by a dash. Resolved by a measured ladder
|
||
# (see _classify_desc) that matches the LLM baseline over the full corpus.
|
||
_BBM_PLUMBING_RE = re.compile(r"plumbing\s*[-–—]")
|
||
# Ladder rungs for a BBM plumbing row, checked in order:
|
||
# (a) emergency/reactive -> Reactive
|
||
_BBM_PLUMBING_REACTIVE_WORD_RE = _kw("reactive", "emergency")
|
||
# (b) technician -> PM
|
||
_BBM_PLUMBING_TECH_RE = _kw("technician")
|
||
# (c) water heater / backflow / water fountain -> PM (before (d), so a
|
||
# 'water heater repair' hits PM here and not Reactive on 'repair')
|
||
_BBM_PLUMBING_PM_RE = _kw("water heater", "backflow", "water fountain")
|
||
# (d) clog/unclog/leak/repair/sewer/drain -> Reactive ('clog'/'leak' prefixes
|
||
# also catch 'clogged'/'leaking')
|
||
_BBM_PLUMBING_REACTIVE_RE = _kw("clog", "unclog", "leak", "repair", "sewer", "drain")
|
||
# (e) else (project, install, misc) -> PM
|
||
# Plumbing-specific qualifiers -- these indicate Plumbing - Reactive on their
|
||
# OWN, without the word 'plumbing' present ('CLOGGED PIT AUGER' -> Reactive).
|
||
_PLUMBING_SPECIFIC_RE = _kw(
|
||
"clog",
|
||
"unclog",
|
||
"sewer",
|
||
"drain",
|
||
"grease trap",
|
||
"jetter",
|
||
"toilet",
|
||
"faucet",
|
||
"urinal",
|
||
)
|
||
# Generic reactive qualifiers -- these require the word 'plumbing' to be present
|
||
# before they route to Plumbing - Reactive.
|
||
_PLUMBING_GENERIC_RE = _kw(
|
||
"reactive",
|
||
"emergency",
|
||
"repair",
|
||
"leak",
|
||
"flood",
|
||
"water line",
|
||
"pipe",
|
||
)
|
||
_ELECTRICAL_RE = _kw(
|
||
"electrical",
|
||
"lighting",
|
||
"ballast",
|
||
"outlet",
|
||
"circuit",
|
||
"panel",
|
||
"generator",
|
||
"transformer",
|
||
"conduit",
|
||
)
|
||
_HVAC_RE = _kw(
|
||
"hvac",
|
||
"heating",
|
||
"cooling",
|
||
"air conditioning",
|
||
"rtu",
|
||
"ahu",
|
||
"vav",
|
||
"refrigerant",
|
||
"thermostat",
|
||
"ductwork",
|
||
)
|
||
_DOCK_DOORS_RE = _kw(
|
||
"dock door",
|
||
"dock leveler",
|
||
"dock plate",
|
||
"dock seal",
|
||
"dock bumper",
|
||
)
|
||
_DOORS_RE = _kw("door", "overhead door", "roll-up", "automatic door", "access door")
|
||
_SIGNAGE_RE = _kw("sign", "banner", "wayfinding", "marquee", "directional")
|
||
_CARPENTRY_RE = _kw("carpentry", "cabinet", "millwork", "trim", "shelving", "framing")
|
||
_FENCE_RE = _kw("fence", "fencing", "bollard")
|
||
_GATE_RE = _kw("gate")
|
||
_CONVEYANCE_RE = _kw("conveyor", "conveyance", "mhe", "material handling", "sortation")
|
||
_PAINTING_RE = _kw("paint", "painting", "primer", "coating", "touch-up")
|
||
_FLOORING_RE = _kw("floor", "tile", "carpet", "epoxy", "polishing")
|
||
_JANITORIAL_RE = _kw(
|
||
"janitorial", "cleaning", "custodial", "pressure wash", "power wash"
|
||
)
|
||
_FIRE_RE = _kw("fire", "sprinkler", "extinguisher", "fire alarm", "suppression")
|
||
_LANDSCAPING_RE = _kw("landscape", "lawn", "tree", "yard", "mowing", "irrigation")
|
||
_ROOFING_RE = _kw("roof", "roofing", "gutter", "downspout")
|
||
# Security/Locksmith: 'lock' and 'key' match as EXACT whole words only (so
|
||
# 'Locker' is not 'lock' and 'keyboard' is not 'key'); the multi-word terms keep
|
||
# the leading-boundary/prefix behavior. The phrase 'key box' is stripped before
|
||
# this test (see _classify_desc) so a Key Box fixture never triggers Locksmith.
|
||
_SECURITY_RE = re.compile(r"\b(?:lock|key)\b|\b(?:access control|camera|security|cctv)")
|
||
_SNOW_RE = _kw("snow", "ice", "salt", "de-ice", "plow")
|
||
_EMERGENCY_RE = _kw("emergency")
|
||
# General Building catch-all sub-shapes.
|
||
_HANDYMAN_RE = re.compile(r"\bhandyman\b")
|
||
_PROJECT_RE = re.compile(r"\bproject\b")
|
||
|
||
TRADE_UPLIFT = "PO Uplift"
|
||
TRADE_GENERAL_BUILDING = "General Building"
|
||
|
||
# Simple (single-regex) trades, in strict priority order. The compound and
|
||
# exception-bearing trades (Plumbing, Electrical, Fencing/Gates, the General
|
||
# Building family, PO Uplift) are handled inline in _classify_desc.
|
||
_SIMPLE_TRADES = (
|
||
(_HVAC_RE, "HVAC"),
|
||
(_DOCK_DOORS_RE, "Dock Doors"),
|
||
(_DOORS_RE, "Doors"),
|
||
(_SIGNAGE_RE, "Signage"),
|
||
(_CARPENTRY_RE, "Carpentry"),
|
||
)
|
||
# Simple trades that follow Fencing/Gates in the priority table. Security and
|
||
# Snow are handled explicitly after this loop (Security needs the 'key box'
|
||
# exclusion applied to its match target, so it can't share the generic loop).
|
||
_SIMPLE_TRADES_TAIL = (
|
||
(_CONVEYANCE_RE, "Conveyance/MHE"),
|
||
(_PAINTING_RE, "Painting"),
|
||
(_FLOORING_RE, "Flooring"),
|
||
(_JANITORIAL_RE, "Janitorial"),
|
||
(_FIRE_RE, "Fire/Life Safety"),
|
||
(_LANDSCAPING_RE, "Landscaping/Yard"),
|
||
(_ROOFING_RE, "Roofing"),
|
||
)
|
||
|
||
|
||
def _classify_desc(desc):
|
||
"""Classify a single non-empty description into one trade label.
|
||
|
||
Returns a label for any non-empty description -- 'General Building' is the
|
||
catch-all when nothing else matches. Callers must not pass empty/blank
|
||
descriptions (that case is handled upstream so it can return None).
|
||
"""
|
||
low = desc[:_MAX_TRADE].lower()
|
||
stripped = low.strip()
|
||
|
||
# PO Uplift: exactly or primarily "PO Uplift".
|
||
if stripped == "po uplift" or stripped.startswith("po uplift"):
|
||
return TRADE_UPLIFT
|
||
|
||
# BBM-structured 'Plumbing - ...' row: resolved by a measured ladder tuned to
|
||
# the LLM baseline over the full corpus (first rung wins).
|
||
if _BBM_PLUMBING_RE.match(stripped):
|
||
if _BBM_PLUMBING_REACTIVE_WORD_RE.search(low): # (a) emergency/reactive
|
||
return "Plumbing - Reactive"
|
||
if _BBM_PLUMBING_TECH_RE.search(low): # (b) technician
|
||
return "Plumbing - PM"
|
||
if _BBM_PLUMBING_PM_RE.search(low): # (c) heater/backflow/fountain
|
||
return "Plumbing - PM"
|
||
if _BBM_PLUMBING_REACTIVE_RE.search(low): # (d) clog/leak/repair/...
|
||
return "Plumbing - Reactive"
|
||
return "Plumbing - PM" # (e) project/install/misc
|
||
# Plumbing - PM (before Reactive) for non-BBM free text.
|
||
if _PM_RE.search(low):
|
||
return "Plumbing - PM"
|
||
# Plumbing - Reactive: a plumbing-specific qualifier ALONE (clog, drain,
|
||
# toilet, ...), or the word 'plumbing' PLUS a generic reactive qualifier.
|
||
if _PLUMBING_SPECIFIC_RE.search(low):
|
||
return "Plumbing - Reactive"
|
||
if _PLUMBING_RE.search(low) and _PLUMBING_GENERIC_RE.search(low):
|
||
return "Plumbing - Reactive"
|
||
|
||
# Electrical -- but electrical keywords do NOT match in a dock-door context.
|
||
if _ELECTRICAL_RE.search(low) and "dock door" not in low:
|
||
return "Electrical"
|
||
|
||
for regex, label in _SIMPLE_TRADES:
|
||
if regex.search(low):
|
||
return label
|
||
|
||
# Fencing/Gates: fence/fencing/bollard anywhere, or 'gate' but NOT 'dock
|
||
# gate' (the dock-gate occurrences are removed before the gate test).
|
||
if _FENCE_RE.search(low) or _GATE_RE.search(low.replace("dock gate", " ")):
|
||
return "Fencing/Gates"
|
||
|
||
for regex, label in _SIMPLE_TRADES_TAIL:
|
||
if regex.search(low):
|
||
return label
|
||
|
||
# Security/Locksmith: 'lock'/'key' as whole words, with 'key box' stripped
|
||
# first (mirrors the 'dock gate' exclusion) so a Key Box fixture is not one.
|
||
if _SECURITY_RE.search(low.replace("key box", " ")):
|
||
return "Security/Locksmith"
|
||
if _SNOW_RE.search(low):
|
||
return "Snow Removal"
|
||
|
||
# General Building family (catch-alls, in priority order).
|
||
if stripped.startswith("emer") or _EMERGENCY_RE.search(low):
|
||
return "General Building - Emergency"
|
||
if _HANDYMAN_RE.search(low) or "general building technician" in low:
|
||
return "General Building - Handyman"
|
||
if "general building project" in low or _PROJECT_RE.search(low):
|
||
return "General Building - Project"
|
||
# Keyword-less BBM 'General Building - <Fixture> - ...' rows (parking lot,
|
||
# fan, ceiling, television, locker, key box, ...) fall through here to the
|
||
# plain catch-all -- that matches the measured LLM majority for such rows.
|
||
return TRADE_GENERAL_BUILDING
|
||
|
||
|
||
def _to_amount(value):
|
||
"""Coerce a line amount (Decimal | str | int/float | junk) to a Decimal for
|
||
comparison. Anything unparseable becomes Decimal(0) -- defensive, never
|
||
raises, so an item with a bad amount simply cannot win the highest-value
|
||
tie-break."""
|
||
if isinstance(value, bool):
|
||
return Decimal(0)
|
||
if isinstance(value, Decimal):
|
||
return value if value.is_finite() else Decimal(0)
|
||
if isinstance(value, int):
|
||
return Decimal(value)
|
||
if isinstance(value, float):
|
||
try:
|
||
d = Decimal(str(value))
|
||
return d if d.is_finite() else Decimal(0)
|
||
except InvalidOperation:
|
||
return Decimal(0)
|
||
if isinstance(value, str):
|
||
try:
|
||
d = Decimal(value.replace(",", "").strip())
|
||
return d if d.is_finite() else Decimal(0)
|
||
except (InvalidOperation, ValueError):
|
||
return Decimal(0)
|
||
return Decimal(0)
|
||
|
||
|
||
def derive_trade(parsed):
|
||
"""Derive the PO's primary trade label, or None. NEVER raises.
|
||
|
||
Each line item with a usable description is classified. The PO trade is the
|
||
primary non-"PO Uplift" trade; when several distinct non-uplift trades are
|
||
present, the trade of the highest-`amount` non-uplift item wins. If every
|
||
described item is PO Uplift, returns "PO Uplift". With no line items or no
|
||
usable descriptions at all, returns None ("General Building" is only ever
|
||
returned for a description that matched nothing).
|
||
"""
|
||
try:
|
||
if not isinstance(parsed, dict):
|
||
return None
|
||
classified = [] # (trade_label, amount_decimal)
|
||
# Aggregate CPU budget across ALL items: the per-item _MAX_TRADE cap
|
||
# alone still allows _MAX_ITEMS x _MAX_TRADE x ~25 regex passes
|
||
# (~10s full-core, minutes under the 256MB Lambda's CPU throttle ->
|
||
# timeout -> async retry -> DLQ). Real POs total well under 10k chars
|
||
# of description text; stop classifying once the budget is spent and
|
||
# decide from what was classified (sh-security-review
|
||
# PO-DERIVED-AVAIL-001 defense-in-depth).
|
||
budget = _MAX_TRADE_TOTAL
|
||
for item in _iter_items(parsed):
|
||
desc = item.get("description")
|
||
if not isinstance(desc, str) or not desc.strip():
|
||
continue
|
||
if budget <= 0:
|
||
break
|
||
desc = desc[:budget]
|
||
budget -= len(desc)
|
||
classified.append((_classify_desc(desc), _to_amount(item.get("amount"))))
|
||
if not classified:
|
||
return None
|
||
|
||
non_uplift = [(t, a) for (t, a) in classified if t != TRADE_UPLIFT]
|
||
if not non_uplift:
|
||
return TRADE_UPLIFT
|
||
if len({t for (t, _) in non_uplift}) == 1:
|
||
return non_uplift[0][0]
|
||
best_trade, best_amount = non_uplift[0]
|
||
for trade, amount in non_uplift[1:]:
|
||
if amount > best_amount:
|
||
best_trade, best_amount = trade, amount
|
||
return best_trade
|
||
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
||
return None
|
||
|
||
|
||
def derive_all(parsed):
|
||
"""All three derived fields as a dict. NEVER raises: on any failure returns
|
||
an all-None dict so the caller always gets the three keys."""
|
||
try:
|
||
return {
|
||
"site_code": derive_site_code(parsed),
|
||
"trade": derive_trade(parsed),
|
||
"fiscal_year": derive_fiscal_year(parsed),
|
||
}
|
||
except Exception: # noqa: BLE001 -- totality: never raise on the email path
|
||
return {"site_code": None, "trade": None, "fiscal_year": None}
|