"""PO parse-outcome and derived-field-agreement EMF telemetry. Pure module: the emit wrappers below print CloudWatch EMF log lines via the shared ``emf`` writers (stdout only, no PutMetricData API call), so this module makes no AWS call and imports no boto3. The derived-agreement wrappers live here (not in the untouchable derived_fields.py) and are imported by enrichment.py. """ import logging from emf import emit_metric, emit_parse_outcome logger = logging.getLogger() logger.setLevel(logging.INFO) # CloudWatch EMF namespace/metric for the parse-outcome metric (the PO # fallback-rate alarm in cdk/po_stack.py reads the ["ParseMethod"] series). METRIC_NAMESPACE = "Seahaven/PoIngest" # Shadow telemetry for the Python-derived classifier bake. One EMF record per # derived field per email, emitted ONLY on the ai_fallback path (the template # path has no LLM value to compare against). Dimensioned by Field x Agreement # only -- PythonValue/LlmValue/po_number ride along as Logs-Insights-queryable # properties so the cardinality stays fixed at (3 fields x 4 categories). DERIVED_METRIC_NAME = "DerivedFieldAgreement" def _emit_parse_method_metric(method, template_id, reason_code, po_number): """Emit one CloudWatch EMF line recording the parse outcome. Zero-latency (no PutMetricData API call): the extraction path is async and the role already has logs:PutLogEvents. ParseMethod/TemplateId are the only promoted (dimensioned) fields to keep cardinality low; ReasonCode and po_number ride along as Logs-Insights-queryable properties. Two dimension sets are published: ["ParseMethod"] (aggregated across all template ids -- the series the fallback-rate alarm queries) AND ["ParseMethod", "TemplateId"] (per-template breakdown for Logs Insights / dashboards). CloudWatch materializes only the exact dimension sets listed here and does NOT auto-aggregate, so the alarm's single-dimension query would receive no data unless ["ParseMethod"] is emitted explicitly.""" emit_parse_outcome( METRIC_NAMESPACE, method, template_id, reason_code, "po_number", po_number ) def _derived_agreement(llm_value, python_value) -> str | None: """Classify Python-vs-LLM agreement for one derived field. Returns None when both values are None (nothing to compare -- the caller then skips emission). Categories: * ``agree`` -- both non-None and equal after str-strip * ``disagree`` -- both non-None but different * ``llm_null_python_filled``-- LLM None, Python supplied a value * ``python_null`` -- LLM non-None, Python None """ if llm_value is None and python_value is None: return None if llm_value is None: return "llm_null_python_filled" if python_value is None: return "python_null" if str(llm_value).strip() == str(python_value).strip(): return "agree" return "disagree" def _emit_derived_agreement_metric(field, llm_value, python_value, po_number): """Emit one CloudWatch EMF line shadowing the Python-derived classifier against the LLM value for a single derived field (ai_fallback path only). Mirrors ``_emit_parse_method_metric``: zero-latency (no PutMetricData; the role already has logs:PutLogEvents), Field x Agreement the only promoted dimension set (cardinality 3x4). PythonValue/LlmValue/po_number ride along as Logs-Insights-queryable properties so a disagreement can be reviewed by example without inflating metric cardinality. No emission when both values are None -- there is nothing to compare.""" agreement = _derived_agreement(llm_value, python_value) if agreement is None: return emit_metric( METRIC_NAMESPACE, DERIVED_METRIC_NAME, [["Field", "Agreement"]], { "Field": field, "Agreement": agreement, "po_number": po_number or "", # Length-clamped: Python values are regex/enum-bounded by construction, # but the LLM value is schema-unvalidated model output -- a hallucinated # free-text field must not land unbounded in a 2-month log line # (sh-security-review PO-DC-02, confirmed low). "PythonValue": "" if python_value is None else str(python_value)[:64], "LlmValue": "" if llm_value is None else str(llm_value)[:64], }, )