apm-wo-analysis/tests/test_classify.py

431 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Smoke test for the two-axis classifier.
Runs the DETERMINISTIC ``classify()`` (no Haiku, no network, no AWS) over the
canonical sample export and asserts the two-axis model keeps "Other" in the
single digits, plus a handful of known fixtures land in the right bucket. The
Haiku fallback is never exercised here.
Canonical fixture: ``~/Downloads/_documents/Sheet1-1.xlsx`` (347 rows, 13 cols).
If the file is absent the export-driven tests skip. Run with the repo venv:
./.venv/bin/python -m pytest tests/test_classify.py -s -q
CSV fixture (always present, runs in CI): ``tests/fixtures/sample_export.csv``.
"""
import collections
import csv
import re
import sys
from pathlib import Path
import pytest
sys.path.insert(
0, str(Path(__file__).resolve().parent.parent / "lambdas" / "classifier")
)
import classify # noqa: E402
# Column indices in the 13-column export.
COL_WO_STATUS = 6
COL_HOLD_REASON = 7
COL_LAST_COMMENT = 8
FIXTURE = Path.home() / "Downloads" / "_documents" / "Sheet1-1.xlsx"
# Committed CSV fixture — always present, no skipif.
CSV_FIXTURE = Path(__file__).resolve().parent / "fixtures" / "sample_export.csv"
# Max acceptable deterministic "Other" share before any AI. CLAUDE.md: the
# two-axis model lands ~9% before Haiku; we hold the line at single digits.
MAX_OTHER_PCT = 10.0
# Threshold for the committed CSV fixture. Measured deterministic Other% on
# the synthetic fixture: 7.41% (2 of 27 classified rows). Threshold is set
# with headroom but still comfortably single-digit.
CSV_FIXTURE_MAX_OTHER_PCT = 9.0
# ---------------------------------------------------------------------------
# CSV loader helper (yields column-indexed tuples like openpyxl row values)
# ---------------------------------------------------------------------------
def _load_csv_rows(path: Path) -> list[tuple]:
"""Read a 13-column CSV export; return data rows as tuples (header skipped)."""
with path.open(newline="") as fh:
reader = csv.reader(fh)
rows = list(reader)
return [tuple(r) for r in rows[1:]] # drop header row
def test_classification_constants_well_formed():
assert "3rd Escalation" in classify.ESCALATION_CATEGORIES
assert classify.HOLD_TO_CATEGORY["SCHEDULING"] == "Awaiting Scheduling"
# Every escalation category is also action-needed.
assert classify.ESCALATION_CATEGORIES <= classify.ACTION_NEEDED_CATEGORIES
def test_strip_html_unwraps_and_normalises():
assert classify.strip_html(None) == ""
assert classify.strip_html(" ") == ""
assert classify.strip_html("<html>WO schedule confirmed with vendor</html>") == (
"WO schedule confirmed with vendor"
)
# Nested tags, entities, smart quotes, and embedded URLs.
raw = (
"<html><div>1st attempt process for schedule confirmation. "
"Vendor, please confirm ‘Schedule Start Date’ &amp; proceed. "
"https://app.avetta.com/avt-cli/x</div></html>"
)
cleaned = classify.strip_html(raw)
assert "<" not in cleaned and ">" not in cleaned
assert "‘" not in cleaned and "’" not in cleaned # smart quotes gone
assert "'Schedule Start Date'" in cleaned
assert "&amp;" not in cleaned and "&" in cleaned # entity decoded
assert "https://" not in cleaned # URL reduced to a token
def test_known_intents():
# Schedule confirmed (the dominant happy-path comment).
cat, _ = classify.classify(
"IP", "", "<html>WO schedule confirmed with vendor</html>"
)
assert cat == "Schedule Confirmed"
# Cancelled via WO Status, even with a generic comment.
cat, _ = classify.classify(
"RCAN", "", "<html>WO Cancelled, created in error.</html>"
)
assert cat == "Cancelled"
# 3rd-attempt escalation cadence.
cat, _ = classify.classify(
"H", "REPORT", "<html>3rd attempt process for schedule confirmation.</html>"
)
assert cat == "3rd Escalation"
# Vendor no-show.
cat, _ = classify.classify(
"R", "", "<html>Site tech reported vendor was a no show for Friday.</html>"
)
assert cat == "Vendor No-Show"
# Weekly cadence template.
cat, _ = classify.classify(
"IP",
"",
"<html>Weekly WO scheduled. Service reports required EOD Friday.</html>",
)
assert cat == "Weekly WO Scheduled"
# Structured-only fallback: no comment intent fires, REPORT hold decides.
cat, _ = classify.classify("IP", "REPORT", "<html></html>")
assert cat == "Report / Docs Needed"
def test_mismatch_detection():
# Comment claims completion while on a REPORT hold → mismatch surfaced.
cat, mm = classify.classify(
"IP", "REPORT", "<html>Vendor arrived and performed task.</html>"
)
assert cat == "Completed / Pending Close"
assert mm is not None and "REPORT" in mm
# Schedule confirmed while on a SCHEDULING hold → mismatch surfaced.
cat, mm = classify.classify(
"R", "SCHEDULING", "<html>WO schedule confirmed with vendor.</html>"
)
assert cat == "Schedule Confirmed"
assert mm is not None
# Clean case: no contradiction → no mismatch.
_, mm = classify.classify(
"IP", "", "<html>WO schedule confirmed with vendor</html>"
)
assert mm is None
def test_classify_is_offline():
"""classify() must not import boto3 or reach the network."""
import sys as _sys
had_boto3 = "boto3" in _sys.modules
classify.classify("IP", "REPORT", "<html>1st attempt process for report.</html>")
# If boto3 wasn't already loaded, classify() must not have pulled it in.
if not had_boto3:
assert "boto3" not in _sys.modules
@pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}")
def test_other_share_against_real_export(capsys):
import openpyxl
wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True)
rows = list(wb.active.iter_rows(values_only=True))[1:] # drop header
dist = collections.Counter()
mismatches = 0
classified = 0
blank = 0
other_samples = []
for row in rows:
comment = row[COL_LAST_COMMENT]
# Blank-comment rows are excluded from the classified total by design.
if comment is None or str(comment).strip() == "":
blank += 1
continue
classified += 1
category, mismatch = classify.classify(
row[COL_WO_STATUS], row[COL_HOLD_REASON], comment
)
dist[category] += 1
if mismatch:
mismatches += 1
if category == "Other" and len(other_samples) < 20:
other_samples.append(classify.strip_html(comment)[:90])
other = dist["Other"]
other_pct = other * 100.0 / classified if classified else 0.0
with capsys.disabled():
print(f"\n=== APM classifier smoke test: {FIXTURE.name} ===")
print(
f"rows={len(rows)} classified={classified} "
f"blank-excluded={blank} (blank-comment rows excluded from the total)"
)
print("--- category distribution ---")
for cat, count in dist.most_common():
print(f" {count:4d} {count * 100.0 / classified:5.1f}% {cat}")
print(f"--- Other: {other} ({other_pct:.2f}%) ---")
for sample in other_samples:
print(f" [Other] {sample}")
print(f"--- mismatches flagged: {mismatches} ---")
assert classified > 0
assert other_pct <= MAX_OTHER_PCT, (
f"deterministic Other {other_pct:.2f}% exceeds {MAX_OTHER_PCT}% — "
"the two-axis ladder regressed"
)
# Two-axis model should comfortably beat the legacy ~17% Other.
assert other_pct < 17.0
@pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}")
def test_known_fixture_rows_in_export():
"""Anchor on real rows found in the canonical export."""
import openpyxl
wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True)
rows = list(wb.active.iter_rows(values_only=True))[1:]
by_status = collections.defaultdict(list)
schedule_confirmed_row = None
report_completion_mismatch = None
for row in rows:
comment = row[COL_LAST_COMMENT]
if comment is None or str(comment).strip() == "":
continue
text = classify.strip_html(comment)
by_status[row[COL_WO_STATUS]].append(row)
if (
schedule_confirmed_row is None
and "schedule confirmed with vendor" in text.lower()
and not (row[COL_HOLD_REASON] or "").strip()
):
schedule_confirmed_row = row
if (
report_completion_mismatch is None
and (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT"
and "performed task" in text.lower()
):
report_completion_mismatch = row
# A "WO schedule confirmed with vendor" row → Schedule Confirmed.
assert schedule_confirmed_row is not None, "fixture lacks a schedule-confirmed row"
cat, _ = classify.classify(
schedule_confirmed_row[COL_WO_STATUS],
schedule_confirmed_row[COL_HOLD_REASON],
schedule_confirmed_row[COL_LAST_COMMENT],
)
assert cat == "Schedule Confirmed"
# Any RCAN row → Cancelled.
assert "RCAN" in by_status, "fixture lacks an RCAN row"
rcan = by_status["RCAN"][0]
cat, _ = classify.classify(
rcan[COL_WO_STATUS], rcan[COL_HOLD_REASON], rcan[COL_LAST_COMMENT]
)
assert cat == "Cancelled"
# A REPORT-hold row whose comment claims completion → mismatch non-None.
if report_completion_mismatch is not None:
_, mm = classify.classify(
report_completion_mismatch[COL_WO_STATUS],
report_completion_mismatch[COL_HOLD_REASON],
report_completion_mismatch[COL_LAST_COMMENT],
)
assert mm is not None
# ---------------------------------------------------------------------------
# CSV fixture tests — always run (no skipif), so the quality gate fires in CI.
# ---------------------------------------------------------------------------
def test_other_share_against_csv_fixture(capsys):
"""Classification quality gate against the committed synthetic CSV fixture.
Measured deterministic Other%: 7.41% (2/27). Threshold: 9.0%.
This test runs unconditionally in CI.
"""
rows = _load_csv_rows(CSV_FIXTURE)
dist: collections.Counter = collections.Counter()
mismatches = 0
classified = 0
blank = 0
other_samples: list[str] = []
for row in rows:
comment = row[COL_LAST_COMMENT] if len(row) > COL_LAST_COMMENT else ""
# Blank-comment rows are excluded from the classified total by design.
if comment is None or str(comment).strip() == "":
blank += 1
continue
classified += 1
category, mismatch = classify.classify(
row[COL_WO_STATUS], row[COL_HOLD_REASON], comment
)
dist[category] += 1
if mismatch:
mismatches += 1
if category == "Other" and len(other_samples) < 20:
other_samples.append(classify.strip_html(comment)[:90])
other = dist["Other"]
other_pct = other * 100.0 / classified if classified else 0.0
with capsys.disabled():
print(f"\n=== APM classifier smoke test (CSV fixture): {CSV_FIXTURE.name} ===")
print(
f"rows={len(rows)} classified={classified} "
f"blank-excluded={blank} (blank-comment rows excluded from the total)"
)
print("--- category distribution ---")
for cat, count in dist.most_common():
print(f" {count:4d} {count * 100.0 / classified:5.1f}% {cat}")
print(f"--- Other: {other} ({other_pct:.2f}%) ---")
for sample in other_samples:
print(f" [Other] {sample}")
print(f"--- mismatches flagged: {mismatches} ---")
assert classified > 0, "CSV fixture produced no classified rows"
assert other_pct <= CSV_FIXTURE_MAX_OTHER_PCT, (
f"deterministic Other {other_pct:.2f}% exceeds {CSV_FIXTURE_MAX_OTHER_PCT}% — "
"the two-axis ladder regressed against the committed fixture"
)
# Fixture must exercise mismatch detection (WO-1012: REPORT hold + performed task).
assert mismatches >= 1, "CSV fixture should contain at least one mismatch row"
def test_known_rows_in_csv_fixture():
"""Anchor checks on the synthetic CSV fixture — runs unconditionally in CI."""
rows = _load_csv_rows(CSV_FIXTURE)
by_status: collections.defaultdict = collections.defaultdict(list)
schedule_confirmed_row = None
report_completion_mismatch = None
third_esc_rows: list[tuple] = []
cancelled_rows: list[tuple] = []
structured_report_rows: list[tuple] = []
structured_scheduling_rows: list[tuple] = []
for row in rows:
comment = row[COL_LAST_COMMENT] if len(row) > COL_LAST_COMMENT else ""
if comment is None or str(comment).strip() == "":
continue
text = classify.strip_html(comment)
by_status[row[COL_WO_STATUS]].append(row)
if (
schedule_confirmed_row is None
and "schedule confirmed with vendor" in text.lower()
and not (row[COL_HOLD_REASON] or "").strip()
):
schedule_confirmed_row = row
if (
report_completion_mismatch is None
and (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT"
and "performed task" in text.lower()
):
report_completion_mismatch = row
if re.search(r"\b3rd\b.*\battempt\b", text.lower()):
third_esc_rows.append(row)
if row[COL_WO_STATUS] == "RCAN":
cancelled_rows.append(row)
# Structured-only: HTML-wrapped empty comment (strips to "") with REPORT hold.
if (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT" and text == "":
structured_report_rows.append(row)
if (row[COL_HOLD_REASON] or "").strip().upper() == "SCHEDULING" and text == "":
structured_scheduling_rows.append(row)
# "WO schedule confirmed with vendor" → Schedule Confirmed.
assert schedule_confirmed_row is not None, "fixture lacks a schedule-confirmed row"
cat, _ = classify.classify(
schedule_confirmed_row[COL_WO_STATUS],
schedule_confirmed_row[COL_HOLD_REASON],
schedule_confirmed_row[COL_LAST_COMMENT],
)
assert cat == "Schedule Confirmed", f"expected Schedule Confirmed, got {cat!r}"
# RCAN rows → Cancelled.
assert cancelled_rows, "fixture lacks an RCAN row"
cat, _ = classify.classify(
cancelled_rows[0][COL_WO_STATUS],
cancelled_rows[0][COL_HOLD_REASON],
cancelled_rows[0][COL_LAST_COMMENT],
)
assert cat == "Cancelled", f"expected Cancelled, got {cat!r}"
# 3rd Escalation rows are present and classify correctly.
assert third_esc_rows, "fixture lacks a 3rd-escalation row"
cat, _ = classify.classify(
third_esc_rows[0][COL_WO_STATUS],
third_esc_rows[0][COL_HOLD_REASON],
third_esc_rows[0][COL_LAST_COMMENT],
)
assert cat == "3rd Escalation", f"expected 3rd Escalation, got {cat!r}"
# Structured-only REPORT rows → Report / Docs Needed.
assert structured_report_rows, "fixture lacks a structured-only REPORT hold row"
cat, _ = classify.classify(
structured_report_rows[0][COL_WO_STATUS],
structured_report_rows[0][COL_HOLD_REASON],
structured_report_rows[0][COL_LAST_COMMENT],
)
assert cat == "Report / Docs Needed", f"expected Report / Docs Needed, got {cat!r}"
# Structured-only SCHEDULING rows → Awaiting Scheduling.
assert structured_scheduling_rows, "fixture lacks a structured-only SCHEDULING row"
cat, _ = classify.classify(
structured_scheduling_rows[0][COL_WO_STATUS],
structured_scheduling_rows[0][COL_HOLD_REASON],
structured_scheduling_rows[0][COL_LAST_COMMENT],
)
assert cat == "Awaiting Scheduling", f"expected Awaiting Scheduling, got {cat!r}"
# REPORT-hold row with completion comment → mismatch surfaced.
assert report_completion_mismatch is not None, (
"fixture lacks a REPORT-hold row with a completion comment (mismatch case)"
)
_, mm = classify.classify(
report_completion_mismatch[COL_WO_STATUS],
report_completion_mismatch[COL_HOLD_REASON],
report_completion_mismatch[COL_LAST_COMMENT],
)
assert mm is not None, "expected a mismatch reason for WO-1012 but got None"
assert "REPORT" in mm, f"mismatch reason should mention REPORT hold: {mm!r}"