"""Smoke test for the two-axis classifier. Runs the DETERMINISTIC ``classify()`` (no Haiku, no network, no AWS) over the canonical sample export and asserts the two-axis model keeps "Other" in the single digits, plus a handful of known fixtures land in the right bucket. The Haiku fallback is never exercised here. Canonical fixture: ``~/Downloads/_documents/Sheet1-1.xlsx`` (347 rows, 13 cols). If the file is absent the export-driven tests skip. Run with the repo venv: ./.venv/bin/python -m pytest tests/test_classify.py -s -q CSV fixture (always present, runs in CI): ``tests/fixtures/sample_export.csv``. """ import collections import csv import re import sys from pathlib import Path import pytest sys.path.insert( 0, str(Path(__file__).resolve().parent.parent / "lambdas" / "classifier") ) import classify # noqa: E402 # Column indices in the 13-column export. COL_WO_STATUS = 6 COL_HOLD_REASON = 7 COL_LAST_COMMENT = 8 FIXTURE = Path.home() / "Downloads" / "_documents" / "Sheet1-1.xlsx" # Committed CSV fixture — always present, no skipif. CSV_FIXTURE = Path(__file__).resolve().parent / "fixtures" / "sample_export.csv" # Max acceptable deterministic "Other" share before any AI. CLAUDE.md: the # two-axis model lands ~9% before Haiku; we hold the line at single digits. MAX_OTHER_PCT = 10.0 # Threshold for the committed CSV fixture. Measured deterministic Other% on # the synthetic fixture: 7.41% (2 of 27 classified rows). Threshold is set # with headroom but still comfortably single-digit. CSV_FIXTURE_MAX_OTHER_PCT = 9.0 # --------------------------------------------------------------------------- # CSV loader helper (yields column-indexed tuples like openpyxl row values) # --------------------------------------------------------------------------- def _load_csv_rows(path: Path) -> list[tuple]: """Read a 13-column CSV export; return data rows as tuples (header skipped).""" with path.open(newline="") as fh: reader = csv.reader(fh) rows = list(reader) return [tuple(r) for r in rows[1:]] # drop header row def test_classification_constants_well_formed(): assert "3rd Escalation" in classify.ESCALATION_CATEGORIES assert classify.HOLD_TO_CATEGORY["SCHEDULING"] == "Awaiting Scheduling" # Every escalation category is also action-needed. assert classify.ESCALATION_CATEGORIES <= classify.ACTION_NEEDED_CATEGORIES def test_strip_html_unwraps_and_normalises(): assert classify.strip_html(None) == "" assert classify.strip_html(" ") == "" assert classify.strip_html("WO schedule confirmed with vendor") == ( "WO schedule confirmed with vendor" ) # Nested tags, entities, smart quotes, and embedded URLs. raw = ( "
1st attempt process for schedule confirmation. " "Vendor, please confirm ‘Schedule Start Date’ & proceed. " "https://app.avetta.com/avt-cli/x
" ) cleaned = classify.strip_html(raw) assert "<" not in cleaned and ">" not in cleaned assert "‘" not in cleaned and "’" not in cleaned # smart quotes gone assert "'Schedule Start Date'" in cleaned assert "&" not in cleaned and "&" in cleaned # entity decoded assert "https://" not in cleaned # URL reduced to a token def test_known_intents(): # Schedule confirmed (the dominant happy-path comment). cat, _ = classify.classify( "IP", "", "WO schedule confirmed with vendor" ) assert cat == "Schedule Confirmed" # Cancelled via WO Status, even with a generic comment. cat, _ = classify.classify( "RCAN", "", "WO Cancelled, created in error." ) assert cat == "Cancelled" # 3rd-attempt escalation cadence. cat, _ = classify.classify( "H", "REPORT", "3rd attempt process for schedule confirmation." ) assert cat == "3rd Escalation" # Vendor no-show. cat, _ = classify.classify( "R", "", "Site tech reported vendor was a no show for Friday." ) assert cat == "Vendor No-Show" # Weekly cadence template. cat, _ = classify.classify( "IP", "", "Weekly WO scheduled. Service reports required EOD Friday.", ) assert cat == "Weekly WO Scheduled" # Structured-only fallback: no comment intent fires, REPORT hold decides. cat, _ = classify.classify("IP", "REPORT", "") assert cat == "Report / Docs Needed" def test_mismatch_detection(): # Comment claims completion while on a REPORT hold → mismatch surfaced. cat, mm = classify.classify( "IP", "REPORT", "Vendor arrived and performed task." ) assert cat == "Completed / Pending Close" assert mm is not None and "REPORT" in mm # Schedule confirmed while on a SCHEDULING hold → mismatch surfaced. cat, mm = classify.classify( "R", "SCHEDULING", "WO schedule confirmed with vendor." ) assert cat == "Schedule Confirmed" assert mm is not None # Clean case: no contradiction → no mismatch. _, mm = classify.classify( "IP", "", "WO schedule confirmed with vendor" ) assert mm is None def test_classify_is_offline(): """classify() must not import boto3 or reach the network.""" import sys as _sys had_boto3 = "boto3" in _sys.modules classify.classify("IP", "REPORT", "1st attempt process for report.") # If boto3 wasn't already loaded, classify() must not have pulled it in. if not had_boto3: assert "boto3" not in _sys.modules @pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}") def test_other_share_against_real_export(capsys): import openpyxl wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True) rows = list(wb.active.iter_rows(values_only=True))[1:] # drop header dist = collections.Counter() mismatches = 0 classified = 0 blank = 0 other_samples = [] for row in rows: comment = row[COL_LAST_COMMENT] # Blank-comment rows are excluded from the classified total by design. if comment is None or str(comment).strip() == "": blank += 1 continue classified += 1 category, mismatch = classify.classify( row[COL_WO_STATUS], row[COL_HOLD_REASON], comment ) dist[category] += 1 if mismatch: mismatches += 1 if category == "Other" and len(other_samples) < 20: other_samples.append(classify.strip_html(comment)[:90]) other = dist["Other"] other_pct = other * 100.0 / classified if classified else 0.0 with capsys.disabled(): print(f"\n=== APM classifier smoke test: {FIXTURE.name} ===") print( f"rows={len(rows)} classified={classified} " f"blank-excluded={blank} (blank-comment rows excluded from the total)" ) print("--- category distribution ---") for cat, count in dist.most_common(): print(f" {count:4d} {count * 100.0 / classified:5.1f}% {cat}") print(f"--- Other: {other} ({other_pct:.2f}%) ---") for sample in other_samples: print(f" [Other] {sample}") print(f"--- mismatches flagged: {mismatches} ---") assert classified > 0 assert other_pct <= MAX_OTHER_PCT, ( f"deterministic Other {other_pct:.2f}% exceeds {MAX_OTHER_PCT}% — " "the two-axis ladder regressed" ) # Two-axis model should comfortably beat the legacy ~17% Other. assert other_pct < 17.0 @pytest.mark.skipif(not FIXTURE.exists(), reason=f"sample export not found: {FIXTURE}") def test_known_fixture_rows_in_export(): """Anchor on real rows found in the canonical export.""" import openpyxl wb = openpyxl.load_workbook(FIXTURE, read_only=True, data_only=True) rows = list(wb.active.iter_rows(values_only=True))[1:] by_status = collections.defaultdict(list) schedule_confirmed_row = None report_completion_mismatch = None for row in rows: comment = row[COL_LAST_COMMENT] if comment is None or str(comment).strip() == "": continue text = classify.strip_html(comment) by_status[row[COL_WO_STATUS]].append(row) if ( schedule_confirmed_row is None and "schedule confirmed with vendor" in text.lower() and not (row[COL_HOLD_REASON] or "").strip() ): schedule_confirmed_row = row if ( report_completion_mismatch is None and (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT" and "performed task" in text.lower() ): report_completion_mismatch = row # A "WO schedule confirmed with vendor" row → Schedule Confirmed. assert schedule_confirmed_row is not None, "fixture lacks a schedule-confirmed row" cat, _ = classify.classify( schedule_confirmed_row[COL_WO_STATUS], schedule_confirmed_row[COL_HOLD_REASON], schedule_confirmed_row[COL_LAST_COMMENT], ) assert cat == "Schedule Confirmed" # Any RCAN row → Cancelled. assert "RCAN" in by_status, "fixture lacks an RCAN row" rcan = by_status["RCAN"][0] cat, _ = classify.classify( rcan[COL_WO_STATUS], rcan[COL_HOLD_REASON], rcan[COL_LAST_COMMENT] ) assert cat == "Cancelled" # A REPORT-hold row whose comment claims completion → mismatch non-None. if report_completion_mismatch is not None: _, mm = classify.classify( report_completion_mismatch[COL_WO_STATUS], report_completion_mismatch[COL_HOLD_REASON], report_completion_mismatch[COL_LAST_COMMENT], ) assert mm is not None # --------------------------------------------------------------------------- # CSV fixture tests — always run (no skipif), so the quality gate fires in CI. # --------------------------------------------------------------------------- def test_other_share_against_csv_fixture(capsys): """Classification quality gate against the committed synthetic CSV fixture. Measured deterministic Other%: 7.41% (2/27). Threshold: 9.0%. This test runs unconditionally in CI. """ rows = _load_csv_rows(CSV_FIXTURE) dist: collections.Counter = collections.Counter() mismatches = 0 classified = 0 blank = 0 other_samples: list[str] = [] for row in rows: comment = row[COL_LAST_COMMENT] if len(row) > COL_LAST_COMMENT else "" # Blank-comment rows are excluded from the classified total by design. if comment is None or str(comment).strip() == "": blank += 1 continue classified += 1 category, mismatch = classify.classify( row[COL_WO_STATUS], row[COL_HOLD_REASON], comment ) dist[category] += 1 if mismatch: mismatches += 1 if category == "Other" and len(other_samples) < 20: other_samples.append(classify.strip_html(comment)[:90]) other = dist["Other"] other_pct = other * 100.0 / classified if classified else 0.0 with capsys.disabled(): print(f"\n=== APM classifier smoke test (CSV fixture): {CSV_FIXTURE.name} ===") print( f"rows={len(rows)} classified={classified} " f"blank-excluded={blank} (blank-comment rows excluded from the total)" ) print("--- category distribution ---") for cat, count in dist.most_common(): print(f" {count:4d} {count * 100.0 / classified:5.1f}% {cat}") print(f"--- Other: {other} ({other_pct:.2f}%) ---") for sample in other_samples: print(f" [Other] {sample}") print(f"--- mismatches flagged: {mismatches} ---") assert classified > 0, "CSV fixture produced no classified rows" assert other_pct <= CSV_FIXTURE_MAX_OTHER_PCT, ( f"deterministic Other {other_pct:.2f}% exceeds {CSV_FIXTURE_MAX_OTHER_PCT}% — " "the two-axis ladder regressed against the committed fixture" ) # Fixture must exercise mismatch detection (WO-1012: REPORT hold + performed task). assert mismatches >= 1, "CSV fixture should contain at least one mismatch row" def test_known_rows_in_csv_fixture(): """Anchor checks on the synthetic CSV fixture — runs unconditionally in CI.""" rows = _load_csv_rows(CSV_FIXTURE) by_status: collections.defaultdict = collections.defaultdict(list) schedule_confirmed_row = None report_completion_mismatch = None third_esc_rows: list[tuple] = [] cancelled_rows: list[tuple] = [] structured_report_rows: list[tuple] = [] structured_scheduling_rows: list[tuple] = [] for row in rows: comment = row[COL_LAST_COMMENT] if len(row) > COL_LAST_COMMENT else "" if comment is None or str(comment).strip() == "": continue text = classify.strip_html(comment) by_status[row[COL_WO_STATUS]].append(row) if ( schedule_confirmed_row is None and "schedule confirmed with vendor" in text.lower() and not (row[COL_HOLD_REASON] or "").strip() ): schedule_confirmed_row = row if ( report_completion_mismatch is None and (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT" and "performed task" in text.lower() ): report_completion_mismatch = row if re.search(r"\b3rd\b.*\battempt\b", text.lower()): third_esc_rows.append(row) if row[COL_WO_STATUS] == "RCAN": cancelled_rows.append(row) # Structured-only: HTML-wrapped empty comment (strips to "") with REPORT hold. if (row[COL_HOLD_REASON] or "").strip().upper() == "REPORT" and text == "": structured_report_rows.append(row) if (row[COL_HOLD_REASON] or "").strip().upper() == "SCHEDULING" and text == "": structured_scheduling_rows.append(row) # "WO schedule confirmed with vendor" → Schedule Confirmed. assert schedule_confirmed_row is not None, "fixture lacks a schedule-confirmed row" cat, _ = classify.classify( schedule_confirmed_row[COL_WO_STATUS], schedule_confirmed_row[COL_HOLD_REASON], schedule_confirmed_row[COL_LAST_COMMENT], ) assert cat == "Schedule Confirmed", f"expected Schedule Confirmed, got {cat!r}" # RCAN rows → Cancelled. assert cancelled_rows, "fixture lacks an RCAN row" cat, _ = classify.classify( cancelled_rows[0][COL_WO_STATUS], cancelled_rows[0][COL_HOLD_REASON], cancelled_rows[0][COL_LAST_COMMENT], ) assert cat == "Cancelled", f"expected Cancelled, got {cat!r}" # 3rd Escalation rows are present and classify correctly. assert third_esc_rows, "fixture lacks a 3rd-escalation row" cat, _ = classify.classify( third_esc_rows[0][COL_WO_STATUS], third_esc_rows[0][COL_HOLD_REASON], third_esc_rows[0][COL_LAST_COMMENT], ) assert cat == "3rd Escalation", f"expected 3rd Escalation, got {cat!r}" # Structured-only REPORT rows → Report / Docs Needed. assert structured_report_rows, "fixture lacks a structured-only REPORT hold row" cat, _ = classify.classify( structured_report_rows[0][COL_WO_STATUS], structured_report_rows[0][COL_HOLD_REASON], structured_report_rows[0][COL_LAST_COMMENT], ) assert cat == "Report / Docs Needed", f"expected Report / Docs Needed, got {cat!r}" # Structured-only SCHEDULING rows → Awaiting Scheduling. assert structured_scheduling_rows, "fixture lacks a structured-only SCHEDULING row" cat, _ = classify.classify( structured_scheduling_rows[0][COL_WO_STATUS], structured_scheduling_rows[0][COL_HOLD_REASON], structured_scheduling_rows[0][COL_LAST_COMMENT], ) assert cat == "Awaiting Scheduling", f"expected Awaiting Scheduling, got {cat!r}" # REPORT-hold row with completion comment → mismatch surfaced. assert report_completion_mismatch is not None, ( "fixture lacks a REPORT-hold row with a completion comment (mismatch case)" ) _, mm = classify.classify( report_completion_mismatch[COL_WO_STATUS], report_completion_mismatch[COL_HOLD_REASON], report_completion_mismatch[COL_LAST_COMMENT], ) assert mm is not None, "expected a mismatch reason for WO-1012 but got None" assert "REPORT" in mm, f"mismatch reason should mention REPORT hold: {mm!r}"