mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-10-01 14:43:13 +00:00
97 lines
4.3 KiB
Python
97 lines
4.3 KiB
Python
|
|
"""PO parse-outcome and derived-field-agreement EMF telemetry.
|
||
|
|
|
||
|
|
Pure module: the emit wrappers below print CloudWatch EMF log lines via the
|
||
|
|
shared ``emf`` writers (stdout only, no PutMetricData API call), so this module
|
||
|
|
makes no AWS call and imports no boto3. The derived-agreement wrappers live here
|
||
|
|
(not in the untouchable derived_fields.py) and are imported by enrichment.py.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import logging
|
||
|
|
|
||
|
|
from emf import emit_metric, emit_parse_outcome
|
||
|
|
|
||
|
|
logger = logging.getLogger()
|
||
|
|
logger.setLevel(logging.INFO)
|
||
|
|
|
||
|
|
# CloudWatch EMF namespace/metric for the parse-outcome metric (the PO
|
||
|
|
# fallback-rate alarm in cdk/po_stack.py reads the ["ParseMethod"] series).
|
||
|
|
METRIC_NAMESPACE = "Seahaven/PoIngest"
|
||
|
|
|
||
|
|
# Shadow telemetry for the Python-derived classifier bake. One EMF record per
|
||
|
|
# derived field per email, emitted ONLY on the ai_fallback path (the template
|
||
|
|
# path has no LLM value to compare against). Dimensioned by Field x Agreement
|
||
|
|
# only -- PythonValue/LlmValue/po_number ride along as Logs-Insights-queryable
|
||
|
|
# properties so the cardinality stays fixed at (3 fields x 4 categories).
|
||
|
|
DERIVED_METRIC_NAME = "DerivedFieldAgreement"
|
||
|
|
|
||
|
|
|
||
|
|
def _emit_parse_method_metric(method, template_id, reason_code, po_number):
|
||
|
|
"""Emit one CloudWatch EMF line recording the parse outcome.
|
||
|
|
|
||
|
|
Zero-latency (no PutMetricData API call): the extraction path is async and
|
||
|
|
the role already has logs:PutLogEvents. ParseMethod/TemplateId are the only
|
||
|
|
promoted (dimensioned) fields to keep cardinality low; ReasonCode and
|
||
|
|
po_number ride along as Logs-Insights-queryable properties.
|
||
|
|
|
||
|
|
Two dimension sets are published: ["ParseMethod"] (aggregated across all
|
||
|
|
template ids -- the series the fallback-rate alarm queries) AND
|
||
|
|
["ParseMethod", "TemplateId"] (per-template breakdown for Logs Insights /
|
||
|
|
dashboards). CloudWatch materializes only the exact dimension sets listed
|
||
|
|
here and does NOT auto-aggregate, so the alarm's single-dimension query
|
||
|
|
would receive no data unless ["ParseMethod"] is emitted explicitly."""
|
||
|
|
emit_parse_outcome(
|
||
|
|
METRIC_NAMESPACE, method, template_id, reason_code, "po_number", po_number
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def _derived_agreement(llm_value, python_value) -> str | None:
|
||
|
|
"""Classify Python-vs-LLM agreement for one derived field.
|
||
|
|
|
||
|
|
Returns None when both values are None (nothing to compare -- the caller
|
||
|
|
then skips emission). Categories:
|
||
|
|
* ``agree`` -- both non-None and equal after str-strip
|
||
|
|
* ``disagree`` -- both non-None but different
|
||
|
|
* ``llm_null_python_filled``-- LLM None, Python supplied a value
|
||
|
|
* ``python_null`` -- LLM non-None, Python None
|
||
|
|
"""
|
||
|
|
if llm_value is None and python_value is None:
|
||
|
|
return None
|
||
|
|
if llm_value is None:
|
||
|
|
return "llm_null_python_filled"
|
||
|
|
if python_value is None:
|
||
|
|
return "python_null"
|
||
|
|
if str(llm_value).strip() == str(python_value).strip():
|
||
|
|
return "agree"
|
||
|
|
return "disagree"
|
||
|
|
|
||
|
|
|
||
|
|
def _emit_derived_agreement_metric(field, llm_value, python_value, po_number):
|
||
|
|
"""Emit one CloudWatch EMF line shadowing the Python-derived classifier
|
||
|
|
against the LLM value for a single derived field (ai_fallback path only).
|
||
|
|
|
||
|
|
Mirrors ``_emit_parse_method_metric``: zero-latency (no PutMetricData; the
|
||
|
|
role already has logs:PutLogEvents), Field x Agreement the only promoted
|
||
|
|
dimension set (cardinality 3x4). PythonValue/LlmValue/po_number ride along
|
||
|
|
as Logs-Insights-queryable properties so a disagreement can be reviewed by
|
||
|
|
example without inflating metric cardinality. No emission when both values
|
||
|
|
are None -- there is nothing to compare."""
|
||
|
|
agreement = _derived_agreement(llm_value, python_value)
|
||
|
|
if agreement is None:
|
||
|
|
return
|
||
|
|
emit_metric(
|
||
|
|
METRIC_NAMESPACE,
|
||
|
|
DERIVED_METRIC_NAME,
|
||
|
|
[["Field", "Agreement"]],
|
||
|
|
{
|
||
|
|
"Field": field,
|
||
|
|
"Agreement": agreement,
|
||
|
|
"po_number": po_number or "",
|
||
|
|
# Length-clamped: Python values are regex/enum-bounded by construction,
|
||
|
|
# but the LLM value is schema-unvalidated model output -- a hallucinated
|
||
|
|
# free-text field must not land unbounded in a 2-month log line
|
||
|
|
# (sh-security-review PO-DC-02, confirmed low).
|
||
|
|
"PythonValue": "" if python_value is None else str(python_value)[:64],
|
||
|
|
"LlmValue": "" if llm_value is None else str(llm_value)[:64],
|
||
|
|
},
|
||
|
|
)
|