mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-10-01 13:33:13 +00:00
Some checks are pending
Deploy / deploy (push) Waiting to run
Both email-processor God-handlers split along the seams that already work in the flat-sibling pattern established by lambdas/shared/, so bare-name imports keep working under the existing bundling glob. PO (5-way split): handler.py keeps only the event loop, fail-closed auth, and email_type routing. extraction.py holds extract_with_claude and _EMAIL_TAG_RE, importing EXTRACTION_PROMPT from prompts.py and parse_raw_email from shared/email_parsing.py rather than recreating a PO-local copy. enrichment.py is a pure code move of enrich_parsed and pad_zip (PO-only; WO has no enrichment stage) with zero behavior change. telemetry.py holds the EMF ParseMethod emit wrappers. persistence.py holds _write_fields/_merge_update/save_*, collapsing the byte-identical save_new_po/save_revision bodies into one _save_merge helper that both now call through, preserving the sticky Cancelled ConditionExpression guard for both callers; save_cancellation stays separate. WO (5 concerns, no enrichment stage): the handler loop keeps validate_ai_fallback and the re.fullmatch(r"[0-9]+", work_order_id) key guard ahead of both save_work_order and save_event, since the guard protects the DynamoDB partition key and the '#'-delimited comment_id range-key segment. _header_date_iso and comment_id determinism stay colocated with persistence.py's save_event for the retry-idempotent event_id key. EXTRACTION_PROMPT (PO) moves to prompts.py with cross-reference headers to derived_fields.py's authoritative trade/site/fiscal rule tables; handler.py re-exports it (from prompts import EXTRACTION_PROMPT) since four tests dereference handler.EXTRACTION_ PROMPT directly. WO's prompt moves the same way. I/O modules (extraction.py's bedrock client, persistence.py's dynamodb resource, handler.py's s3 client) get lazy cached boto3 accessors; pure modules (enrichment.py, prompts.py, telemetry.py) import no boto3. Test monkeypatch surfaces move to the module that now owns the client (e.g. persistence.dynamodb) everywhere tests patch it, and the moto-before-handler-import ordering in _po_parser_support.py is preserved so the moto-backed suites don't hit real AWS. Behavior-preservation pins, verified with tests: PO still emits ParseMethod=ai_fallback before the Bedrock call, with ai_fallback_rejected as the additive second datapoint on rejection. WO still emits after its gate with mutually-exclusive ai_fallback / ai_fallback_rejected. Shadow DerivedFieldAgreement telemetry stays ai_fallback-only. derived_fields.py is untouched (diff against feature/phase-3-shared-extraction is empty). handler(event, context) signatures and the save_* public contract are unchanged on both pipelines; goldens unchanged. PO_EXPECTED_TOP_LEVEL_MODULES and its WO equivalent in tests/test_bundle_consistency.py are updated for the new sibling modules so the AST bundle-consistency test still fails on an unshipped or uncommented-out sibling.
112 lines
5 KiB
Python
112 lines
5 KiB
Python
"""PO post-parse enrichment (PO-only; WO has no enrichment stage).
|
|
|
|
Adds metadata, promotes nested ship-to fields, canonicalizes numeric types
|
|
across both parse paths, and runs the deterministic derived-field classifiers
|
|
(site_code, trade, fiscal_year) that FILL GAPS but never overwrite an
|
|
LLM-supplied value during the bake. Pure module: no boto3. The derived-agreement
|
|
shadow EMF is emitted via telemetry (ai_fallback path only).
|
|
"""
|
|
|
|
import logging
|
|
from datetime import datetime, timezone
|
|
from decimal import Decimal, InvalidOperation
|
|
|
|
from derived_fields import derive_all
|
|
from telemetry import _emit_derived_agreement_metric
|
|
|
|
logger = logging.getLogger()
|
|
logger.setLevel(logging.INFO)
|
|
|
|
# The three classifier outputs derive_all() computes. Python fills these when
|
|
# the extraction path left them null; on ai_fallback the LLM value (if any)
|
|
# stays authoritative during the bake and Python only shadows it.
|
|
DERIVED_FIELDS = ("site_code", "trade", "fiscal_year")
|
|
|
|
|
|
def pad_zip(zip_code: str | None) -> str | None:
|
|
if not zip_code:
|
|
return zip_code
|
|
clean = zip_code.strip().split("-")[0]
|
|
if clean.isdigit() and len(clean) < 5:
|
|
return clean.zfill(5) + zip_code.strip()[len(clean) :]
|
|
return zip_code
|
|
|
|
|
|
def enrich_parsed(parsed: dict, s3_key: str, email_subject: str, *, parse_method: str):
|
|
"""Add metadata and promote nested fields to top level.
|
|
|
|
``parse_method`` ("template" | "ai_fallback") selects the derived-field
|
|
shadow behavior below: agreement telemetry is emitted only on ai_fallback,
|
|
where an LLM value exists to compare the Python classifier against.
|
|
"""
|
|
now = datetime.now(timezone.utc).isoformat()
|
|
parsed["raw_s3_key"] = s3_key
|
|
parsed["processed_at"] = now
|
|
parsed["data_source"] = "email"
|
|
parsed["email_subject"] = email_subject
|
|
|
|
ship_to = parsed.get("ship_to") or {}
|
|
if ship_to.get("address"):
|
|
parsed["ship_to_raw"] = ship_to["address"]
|
|
if ship_to.get("state"):
|
|
parsed["state"] = ship_to["state"]
|
|
|
|
if ship_to.get("zip"):
|
|
ship_to["zip"] = pad_zip(ship_to["zip"])
|
|
|
|
# Canonical numeric type for quantity/price across BOTH parse paths: the
|
|
# template parser emits Decimal (DynamoDB Number) while EXTRACTION_PROMPT
|
|
# asks the LLM for these two fields as JSON strings (which the Bedrock
|
|
# json.loads leaves as str -> DynamoDB String). Coercing here -- in the
|
|
# SHARED post-stage -- converges the attribute type to Number for
|
|
# equivalent parsed dicts, preserving the two-path parity contract on the
|
|
# purchase-orders table stream. Prompt rewording itself is PR #2 scope.
|
|
# A non-numeric string is left verbatim (still stored, as a String) --
|
|
# dropping it would lose LLM-extracted evidence.
|
|
for item in parsed.get("line_items") or []:
|
|
if not isinstance(item, dict):
|
|
continue
|
|
for field in ("quantity", "price"):
|
|
value = item.get(field)
|
|
if isinstance(value, str):
|
|
try:
|
|
item[field] = Decimal(value.replace(",", "").strip())
|
|
except InvalidOperation:
|
|
pass
|
|
elif isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
# parse_float=Decimal means floats can't occur on the LLM path,
|
|
# but a bare JSON int would slip through as Python int; coerce
|
|
# so both paths emit one canonical Decimal type (cross-review FIX).
|
|
item[field] = Decimal(str(value))
|
|
|
|
# Derived-field classification (site_code, trade, fiscal_year). Python
|
|
# derivation FILLS GAPS on BOTH paths but NEVER OVERWRITES: an LLM-supplied
|
|
# value (only possible on the ai_fallback path) stays authoritative during
|
|
# the bake period. On ai_fallback we additionally emit one shadow EMF record
|
|
# per field comparing the Python value to the LLM value, so agreement can be
|
|
# measured before Python becomes authoritative and the rules are dropped
|
|
# from EXTRACTION_PROMPT (a post-bake follow-up).
|
|
#
|
|
# The whole block is wrapped defensively: derive_all() is total and pure,
|
|
# but this is an S3-async Lambda where any uncaught exception means a retry
|
|
# storm -> DLQ, so no classification/telemetry error may ever fail the
|
|
# invocation.
|
|
try:
|
|
python_vals = derive_all(parsed)
|
|
for field in DERIVED_FIELDS:
|
|
llm_value = parsed.get(field)
|
|
python_value = python_vals.get(field)
|
|
if llm_value is None and python_value is not None:
|
|
# Python fills the gap on both paths.
|
|
parsed[field] = python_value
|
|
# else: a non-None LLM value (ai_fallback only) is kept as-is.
|
|
if parse_method == "ai_fallback":
|
|
_emit_derived_agreement_metric(
|
|
field, llm_value, python_value, parsed.get("po_number")
|
|
)
|
|
except Exception: # noqa: BLE001 - telemetry must never fail the invocation
|
|
logger.exception(
|
|
"derived-field classification/telemetry failed; continuing without it"
|
|
)
|
|
|
|
return parsed
|