apm-wo-analysis/tests/test_classify.py

247 lines
8.7 KiB
Python
Raw Normal View History

"""Smoke test for the two-axis classifier.
Runs the DETERMINISTIC ``classify()`` (no Haiku, no network, no AWS) over the
canonical sample export and asserts the two-axis model keeps "Other" in the
single digits, plus a handful of known fixtures land in the right bucket. The
Haiku fallback is never exercised here.
Canonical fixture: ``~/Downloads/_documents/Sheet1-1.xlsx`` (347 rows, 13 cols).
If the file is absent the export-driven tests skip. Run with the repo venv:
./.venv/bin/python -m pytest tests/test_classify.py -s -q
"""
import collections
import sys
from pathlib import Path
import pytest
sys.path.insert(
0, str(Path(__file__).resolve().parent.parent / "lambdas" / "classifier")
)
import classify # noqa: E402
# Column indices in the 13-column export.
COL_WO_STATUS = 6
COL_HOLD_REASON = 7
COL_LAST_COMMENT = 8
FIXTURE = Path.home() / "Downloads" / "_documents" / "Sheet1-1.xlsx"
# Max acceptable deterministic "Other" share before any AI. CLAUDE.md: the
# two-axis model lands ~9% before Haiku; we hold the line at single digits.
MAX_OTHER_PCT = 10.0
def test_classification_constants_well_formed():
assert "3rd Escalation" in classify.ESCALATION_CATEGORIES
assert classify.HOLD_TO_CATEGORY["SCHEDULING"] == "Awaiting Scheduling"
# Every escalation category is also action-needed.
assert classify.ESCALATION_CATEGORIES <= classify.ACTION_NEEDED_CATEGORIES
def test_strip_html_unwraps_and_normalises():
assert classify.strip_html(None) == ""
assert classify.strip_html(" ") == ""
assert classify.strip_html("<html>WO schedule confirmed with vendor</html>") == (
"WO schedule confirmed with vendor"
)
# Nested tags, entities, smart quotes, and embedded URLs.
raw = (
"<html><div>1st attempt process for schedule confirmation. "
"Vendor, please confirm ‘Schedule Start Date’ &amp; proceed. "
"https://app.avetta.com/avt-cli/x</div></html>"
)
cleaned = classify.strip_html(raw)
assert "<" not in cleaned and ">" not in cleaned
assert "‘" not in cleaned and "’" not in cleaned # smart quotes gone
assert "'Schedule Start Date'" in cleaned
assert "&amp;" not in cleaned and "&" in cleaned # entity decoded
assert "https://" not in cleaned # URL reduced to a token
def test_known_intents():
# Schedule confirmed (the dominant happy-path comment).
cat, _ = classify.classify(
"IP", "", "<html>WO schedule confirmed with vendor</html>"
)
assert cat == "Schedule Confirmed"
# Cancelled via WO Status, even with a generic comment.
cat, _ = classify.classify(
"RCAN", "", "<html>WO Cancelled, created in error.</html>"
)
assert cat == "Cancelled"
# 3rd-attempt escalation cadence.
cat, _ = classify.classify(
"H", "REPORT", "<html>3rd attempt process for schedule confirmation.</html>"
)
assert cat == "3rd Escalation"
# Vendor no-show.
cat, _ = classify.classify(
"R", "", "<html>Site tech reported vendor was a no show for Friday.</html>"
)
assert cat == "Vendor No-Show"
# Weekly cadence template.
cat, _ = classify.classify(
"IP",
"",
"<html>Weekly WO scheduled. Service reports required EOD Friday.</html>",
)
assert cat == "Weekly WO Scheduled"
# Structured-only fallback: no comment intent fires, REPORT hold decides.
cat, _ = classify.classify("IP", "REPORT", "<html></html>")
assert cat == "Report / Docs Needed"
def test_mismatch_detection():
# Comment claims completion while on a REPORT hold → mismatch surfaced.
cat, mm = classify.classify(
"IP", "REPORT", "<html>Vendor arrived and performed task.</html>"
)
assert cat == "Completed / Pending Close"
assert mm is not None and "REPORT" in mm
# Schedule confirmed while on a SCHEDULING hold → mismatch surfaced.
cat, mm = classify.classify(
"R", "SCHEDULING", "<html>WO schedule confirmed with vendor.</html>"
)
assert cat == "Schedule Confirmed"
assert mm is not None
# Clean case: no contradiction → no mismatch.
_, mm = classify.classify(
"IP", "", "<html>WO schedule confirmed with vendor</html>"
)
assert mm is None
def test_classify_is_offline():
"""classify() must not import boto3 or reach the network."""
import sys as _sys
had_boto3 = "boto3" in _sys.modules
classify.classify("IP", "REPORT", "<html>1st attempt process for report.</html>")
# If boto3 wasn't already loaded, classify() must not have pulled it in.
if not had_boto3:
assert "boto3" not in _sys.modules
@pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}")
def test_other_share_against_real_export(capsys):
import openpyxl
wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True)
rows = list(wb.active.iter_rows(values_only=True))[1:] # drop header
dist = collections.Counter()
mismatches = 0
classified = 0
blank = 0
other_samples = []
for row in rows:
comment = row[COL_LAST_COMMENT]
# Blank-comment rows are excluded from the classified total by design.
if comment is None or str(comment).strip() == "":
blank += 1
continue
classified += 1
category, mismatch = classify.classify(
row[COL_WO_STATUS], row[COL_HOLD_REASON], comment
)
dist[category] += 1
if mismatch:
mismatches += 1
if category == "Other" and len(other_samples) < 20:
other_samples.append(classify.strip_html(comment)[:90])
other = dist["Other"]
other_pct = other * 100.0 / classified if classified else 0.0
with capsys.disabled():
print(f"\n=== APM classifier smoke test: {FIXTURE.name} ===")
print(
f"rows={len(rows)} classified={classified} "
f"blank-excluded={blank} (blank-comment rows excluded from the total)"
)
print("--- category distribution ---")
for cat, count in dist.most_common():
print(f" {count:4d} {count * 100.0 / classified:5.1f}% {cat}")
print(f"--- Other: {other} ({other_pct:.2f}%) ---")
for sample in other_samples:
print(f" [Other] {sample}")
print(f"--- mismatches flagged: {mismatches} ---")
assert classified > 0
assert other_pct <= MAX_OTHER_PCT, (
f"deterministic Other {other_pct:.2f}% exceeds {MAX_OTHER_PCT}% — "
"the two-axis ladder regressed"
)
# Two-axis model should comfortably beat the legacy ~17% Other.
assert other_pct < 17.0
@pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}")
def test_known_fixture_rows_in_export():
"""Anchor on real rows found in the canonical export."""
import openpyxl
wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True)
rows = list(wb.active.iter_rows(values_only=True))[1:]
by_status = collections.defaultdict(list)
schedule_confirmed_row = None
report_completion_mismatch = None
for row in rows:
comment = row[COL_LAST_COMMENT]
if comment is None or str(comment).strip() == "":
continue
text = classify.strip_html(comment)
by_status[row[COL_WO_STATUS]].append(row)
if (
schedule_confirmed_row is None
and "schedule confirmed with vendor" in text.lower()
and not (row[COL_HOLD_REASON] or "").strip()
):
schedule_confirmed_row = row
if (
report_completion_mismatch is None
and (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT"
and "performed task" in text.lower()
):
report_completion_mismatch = row
# A "WO schedule confirmed with vendor" row → Schedule Confirmed.
assert schedule_confirmed_row is not None, "fixture lacks a schedule-confirmed row"
cat, _ = classify.classify(
schedule_confirmed_row[COL_WO_STATUS],
schedule_confirmed_row[COL_HOLD_REASON],
schedule_confirmed_row[COL_LAST_COMMENT],
)
assert cat == "Schedule Confirmed"
# Any RCAN row → Cancelled.
assert "RCAN" in by_status, "fixture lacks an RCAN row"
rcan = by_status["RCAN"][0]
cat, _ = classify.classify(
rcan[COL_WO_STATUS], rcan[COL_HOLD_REASON], rcan[COL_LAST_COMMENT]
)
assert cat == "Cancelled"
# A REPORT-hold row whose comment claims completion → mismatch non-None.
if report_completion_mismatch is not None:
_, mm = classify.classify(
report_completion_mismatch[COL_WO_STATUS],
report_completion_mismatch[COL_HOLD_REASON],
report_completion_mismatch[COL_LAST_COMMENT],
)
assert mm is not None