apm-wo-analysis/tests/test_classify.py
Adam Moussa eb724580e0 Add two-axis classifier Lambda and CDK wiring (Phase 2)
Implement the core classification engine and wire it into the pipeline stack.

classify.py: two-axis classifier — HTML-strip, comment-intent regex buckets
(escalations → status inquiry, most-specific first), Hold Reason / WO Status
structured state, comment-vs-state mismatch detector, and a Claude Haiku
fallback (Secrets Manager key) reserved for ambiguous free-text. Exports
ESCALATION_CATEGORIES / ACTION_NEEDED_CATEGORIES.

handler.py: S3-triggered handler — parse xlsx/csv, classify each non-blank
row, write a per-WO Parquet snapshot to analytics/dt=YYYY-MM-DD/ (registers
the Glue partition via awswrangler) and a summary.json for slack-post (Phase 4).

pipeline_stack.py: Glue database, ARM64 Python 3.12 classifier Lambda
(Docker-bundled deps), S3 raw/ notification (.xlsx/.csv), and least-privilege
IAM (read raw/, read-write analytics/, scoped Glue catalog, read Anthropic key).

Smoke-tested against the real export: 347 rows, "Other" at 5.2% (target ~9%),
18 mismatches flagged. 7/7 unit + smoke tests pass; cdk synth green.
2026-05-28 17:09:52 -04:00

246 lines
8.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Smoke test for the two-axis classifier.
Runs the DETERMINISTIC ``classify()`` (no Haiku, no network, no AWS) over the
canonical sample export and asserts the two-axis model keeps "Other" in the
single digits, plus a handful of known fixtures land in the right bucket. The
Haiku fallback is never exercised here.
Canonical fixture: ``~/Downloads/_documents/Sheet1-1.xlsx`` (347 rows, 13 cols).
If the file is absent the export-driven tests skip. Run with the repo venv:
./.venv/bin/python -m pytest tests/test_classify.py -s -q
"""
import collections
import sys
from pathlib import Path
import pytest
sys.path.insert(
0, str(Path(__file__).resolve().parent.parent / "lambdas" / "classifier")
)
import classify # noqa: E402
# Column indices in the 13-column export.
COL_WO_STATUS = 6
COL_HOLD_REASON = 7
COL_LAST_COMMENT = 8
FIXTURE = Path.home() / "Downloads" / "_documents" / "Sheet1-1.xlsx"
# Max acceptable deterministic "Other" share before any AI. CLAUDE.md: the
# two-axis model lands ~9% before Haiku; we hold the line at single digits.
MAX_OTHER_PCT = 10.0
def test_classification_constants_well_formed():
assert "3rd Escalation" in classify.ESCALATION_CATEGORIES
assert classify.HOLD_TO_CATEGORY["SCHEDULING"] == "Awaiting Scheduling"
# Every escalation category is also action-needed.
assert classify.ESCALATION_CATEGORIES <= classify.ACTION_NEEDED_CATEGORIES
def test_strip_html_unwraps_and_normalises():
assert classify.strip_html(None) == ""
assert classify.strip_html(" ") == ""
assert classify.strip_html("<html>WO schedule confirmed with vendor</html>") == (
"WO schedule confirmed with vendor"
)
# Nested tags, entities, smart quotes, and embedded URLs.
raw = (
"<html><div>1st attempt process for schedule confirmation. "
"Vendor, please confirm ‘Schedule Start Date’ &amp; proceed. "
"https://app.avetta.com/avt-cli/x</div></html>"
)
cleaned = classify.strip_html(raw)
assert "<" not in cleaned and ">" not in cleaned
assert "‘" not in cleaned and "’" not in cleaned # smart quotes gone
assert "'Schedule Start Date'" in cleaned
assert "&amp;" not in cleaned and "&" in cleaned # entity decoded
assert "https://" not in cleaned # URL reduced to a token
def test_known_intents():
# Schedule confirmed (the dominant happy-path comment).
cat, _ = classify.classify(
"IP", "", "<html>WO schedule confirmed with vendor</html>"
)
assert cat == "Schedule Confirmed"
# Cancelled via WO Status, even with a generic comment.
cat, _ = classify.classify(
"RCAN", "", "<html>WO Cancelled, created in error.</html>"
)
assert cat == "Cancelled"
# 3rd-attempt escalation cadence.
cat, _ = classify.classify(
"H", "REPORT", "<html>3rd attempt process for schedule confirmation.</html>"
)
assert cat == "3rd Escalation"
# Vendor no-show.
cat, _ = classify.classify(
"R", "", "<html>Site tech reported vendor was a no show for Friday.</html>"
)
assert cat == "Vendor No-Show"
# Weekly cadence template.
cat, _ = classify.classify(
"IP",
"",
"<html>Weekly WO scheduled. Service reports required EOD Friday.</html>",
)
assert cat == "Weekly WO Scheduled"
# Structured-only fallback: no comment intent fires, REPORT hold decides.
cat, _ = classify.classify("IP", "REPORT", "<html></html>")
assert cat == "Report / Docs Needed"
def test_mismatch_detection():
# Comment claims completion while on a REPORT hold → mismatch surfaced.
cat, mm = classify.classify(
"IP", "REPORT", "<html>Vendor arrived and performed task.</html>"
)
assert cat == "Completed / Pending Close"
assert mm is not None and "REPORT" in mm
# Schedule confirmed while on a SCHEDULING hold → mismatch surfaced.
cat, mm = classify.classify(
"R", "SCHEDULING", "<html>WO schedule confirmed with vendor.</html>"
)
assert cat == "Schedule Confirmed"
assert mm is not None
# Clean case: no contradiction → no mismatch.
_, mm = classify.classify(
"IP", "", "<html>WO schedule confirmed with vendor</html>"
)
assert mm is None
def test_classify_is_offline():
"""classify() must not import boto3 or reach the network."""
import sys as _sys
had_boto3 = "boto3" in _sys.modules
classify.classify("IP", "REPORT", "<html>1st attempt process for report.</html>")
# If boto3 wasn't already loaded, classify() must not have pulled it in.
if not had_boto3:
assert "boto3" not in _sys.modules
@pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}")
def test_other_share_against_real_export(capsys):
import openpyxl
wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True)
rows = list(wb.active.iter_rows(values_only=True))[1:] # drop header
dist = collections.Counter()
mismatches = 0
classified = 0
blank = 0
other_samples = []
for row in rows:
comment = row[COL_LAST_COMMENT]
# Blank-comment rows are excluded from the classified total by design.
if comment is None or str(comment).strip() == "":
blank += 1
continue
classified += 1
category, mismatch = classify.classify(
row[COL_WO_STATUS], row[COL_HOLD_REASON], comment
)
dist[category] += 1
if mismatch:
mismatches += 1
if category == "Other" and len(other_samples) < 20:
other_samples.append(classify.strip_html(comment)[:90])
other = dist["Other"]
other_pct = other * 100.0 / classified if classified else 0.0
with capsys.disabled():
print(f"\n=== APM classifier smoke test: {FIXTURE.name} ===")
print(
f"rows={len(rows)} classified={classified} "
f"blank-excluded={blank} (blank-comment rows excluded from the total)"
)
print("--- category distribution ---")
for cat, count in dist.most_common():
print(f" {count:4d} {count * 100.0 / classified:5.1f}% {cat}")
print(f"--- Other: {other} ({other_pct:.2f}%) ---")
for sample in other_samples:
print(f" [Other] {sample}")
print(f"--- mismatches flagged: {mismatches} ---")
assert classified > 0
assert other_pct <= MAX_OTHER_PCT, (
f"deterministic Other {other_pct:.2f}% exceeds {MAX_OTHER_PCT}% — "
"the two-axis ladder regressed"
)
# Two-axis model should comfortably beat the legacy ~17% Other.
assert other_pct < 17.0
@pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}")
def test_known_fixture_rows_in_export():
"""Anchor on real rows found in the canonical export."""
import openpyxl
wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True)
rows = list(wb.active.iter_rows(values_only=True))[1:]
by_status = collections.defaultdict(list)
schedule_confirmed_row = None
report_completion_mismatch = None
for row in rows:
comment = row[COL_LAST_COMMENT]
if comment is None or str(comment).strip() == "":
continue
text = classify.strip_html(comment)
by_status[row[COL_WO_STATUS]].append(row)
if (
schedule_confirmed_row is None
and "schedule confirmed with vendor" in text.lower()
and not (row[COL_HOLD_REASON] or "").strip()
):
schedule_confirmed_row = row
if (
report_completion_mismatch is None
and (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT"
and "performed task" in text.lower()
):
report_completion_mismatch = row
# A "WO schedule confirmed with vendor" row → Schedule Confirmed.
assert schedule_confirmed_row is not None, "fixture lacks a schedule-confirmed row"
cat, _ = classify.classify(
schedule_confirmed_row[COL_WO_STATUS],
schedule_confirmed_row[COL_HOLD_REASON],
schedule_confirmed_row[COL_LAST_COMMENT],
)
assert cat == "Schedule Confirmed"
# Any RCAN row → Cancelled.
assert "RCAN" in by_status, "fixture lacks an RCAN row"
rcan = by_status["RCAN"][0]
cat, _ = classify.classify(
rcan[COL_WO_STATUS], rcan[COL_HOLD_REASON], rcan[COL_LAST_COMMENT]
)
assert cat == "Cancelled"
# A REPORT-hold row whose comment claims completion → mismatch non-None.
if report_completion_mismatch is not None:
_, mm = classify.classify(
report_completion_mismatch[COL_WO_STATUS],
report_completion_mismatch[COL_HOLD_REASON],
report_completion_mismatch[COL_LAST_COMMENT],
)
assert mm is not None