"""Golden-file tests for the deterministic PO template parser. Positives: every scrubbed real sample must parse via the deterministic path and match its checked-in golden exactly (pre-enrichment shape, derived fields null, Decimal money). Negatives: every ai-fallback / adversarial sample must fail closed to the AI fallback (parsed is None). File names are prefixed ``test_po_`` (not the WO suite's ``test_parser`` etc.) because pytest imports test modules by basename when the directories are not packages -- duplicate basenames across the WO and PO suites would collide in one session. """ import re import pytest from _po_parser_support import ( ADVERSARIAL_STEMS, AI_FALLBACK_STEMS, CANCELLATION_STEMS, NEW_PO_STEMS, load_email, load_golden, template_parser, ) try_deterministic_parse = template_parser.try_deterministic_parse @pytest.mark.parametrize("stem", NEW_PO_STEMS) def test_new_po_matches_golden(stem): email_data = load_email("new-po", stem) parsed, method, template_id, reason = try_deterministic_parse(email_data) assert method == "template" assert template_id == "coupa_new_po" assert reason == "ok" assert parsed == load_golden(stem) assert parsed["email_type"] == "new_po" assert parsed["source_system"] == "coupa" # Derived fields stay None on the template path: the shared enrich stage # (LLM-only classification for now) must see identical shapes on both paths. for key in template_parser.DERIVED_KEYS: assert parsed[key] is None @pytest.mark.parametrize("stem", CANCELLATION_STEMS) def test_cancellation_matches_golden(stem): email_data = load_email("cancellation", stem) parsed, method, template_id, reason = try_deterministic_parse(email_data) assert method == "template" assert template_id == "coupa_cancellation" assert reason == "ok" assert parsed == load_golden(stem) # The sticky-Cancelled regression guard: a cancellation subject must NEVER # yield email_type='new_po' (the handler's else-branch would route it to # save_new_po and silently defeat the Cancelled guard). assert parsed["email_type"] == "cancellation" assert parsed["po_number"] @pytest.mark.parametrize("stem", AI_FALLBACK_STEMS) def test_ai_fallback_samples_return_none(stem): email_data = load_email("ai-fallback", stem) parsed, method, _template_id, reason = try_deterministic_parse(email_data) assert parsed is None, f"{stem} should have failed closed, got {parsed}" assert method == "ai_fallback" assert reason != "ok" @pytest.mark.parametrize("stem", ADVERSARIAL_STEMS) def test_adversarial_samples_do_not_raise(stem): # The parser must be total: adversarial input returns a tuple, never raises. result = try_deterministic_parse(load_email("adversarial", stem)) assert isinstance(result, tuple) and len(result) == 4 def test_positive_result_matches_ai_contract_keys(): """The deterministic result must carry EXACTLY the nested key sets the AI fallback produces -- no more, no less -- at every level.""" parsed, *_ = try_deterministic_parse(load_email("new-po", NEW_PO_STEMS[0])) assert set(parsed.keys()) == set(template_parser.CONTRACT_KEYS) assert set(parsed["supplier"].keys()) == set(template_parser.SUPPLIER_KEYS) assert set(parsed["ship_to"].keys()) == set(template_parser.SHIP_TO_KEYS) for item in parsed["line_items"]: assert set(item.keys()) == set(template_parser.LINE_ITEM_KEYS) def test_part_number_bullet_shift_regression(): """The optional 'Part Number' bullet segment (present in new-po-01, absent in new-po-09) must not shift Need By/Category/Account/Period assignment -- segments are label-keyed, never ordinal.""" for stem in ("new-po-01", "new-po-09"): parsed, _m, _t, reason = try_deterministic_parse(load_email("new-po", stem)) assert reason == "ok" item = parsed["line_items"][0] assert re.fullmatch(r"\d{2}/\d{2}/\d{2}", item["need_by"]) assert item["category"] == parsed["coupa_category"] assert re.match(r"^[A-Z0-9]", item["account_code"]) assert item["period"] def test_multiline_extracted_faithfully_before_rejection(): """Multi-line-item emails are rejected honestly: the extractor parses EVERY block (sum(lines)==total holds), then the gate fires multiline_unsupported -- it never silently keeps item 0.""" email_data = load_email("ai-fallback", "new-po-05") candidate = template_parser.extract_new_po(email_data) assert len(candidate["line_items"]) == 3 assert ( sum(item["amount"] for item in candidate["line_items"]) == candidate["total_amount"] ) parsed, method, _t, reason = try_deterministic_parse(email_data) assert parsed is None assert method == "ai_fallback" assert reason == "multiline_unsupported" def test_extractor_exception_fails_closed(monkeypatch): # Any exception inside the extractor must yield None, never a partial parse. monkeypatch.setattr( template_parser, "extract_new_po", lambda *_a, **_k: (_ for _ in ()).throw(RuntimeError("boom")), ) parsed, method, _t, reason = template_parser.try_deterministic_parse( load_email("new-po", NEW_PO_STEMS[0]) ) assert parsed is None assert method == "ai_fallback" assert reason == "extractor_raised" def test_body_none_fails_closed(): email_data = dict(load_email("new-po", NEW_PO_STEMS[0]), body=None) parsed, method, _t, reason = try_deterministic_parse(email_data) assert parsed is None assert method == "ai_fallback" assert reason == "extractor_raised" @pytest.mark.parametrize("stem", NEW_PO_STEMS) def test_line_ending_representations_parse_identically(stem): """The extractor must be representation-agnostic: the body reaches it with CRLF line endings on the wire decode path but LF after normalization -- both must produce byte-identical parses (this exact divergence bit the fixture scrubber's first pass).""" email_data = load_email("new-po", stem) lf_body = email_data["body"].replace("\r\n", "\n").replace("\r", "\n") crlf_body = lf_body.replace("\n", "\r\n") parsed_orig, *_ = try_deterministic_parse(email_data) parsed_lf, _m1, _t1, r1 = try_deterministic_parse(dict(email_data, body=lf_body)) parsed_crlf, _m2, _t2, r2 = try_deterministic_parse( dict(email_data, body=crlf_body) ) assert r1 == r2 == "ok" assert parsed_orig == parsed_lf == parsed_crlf def test_prompt_injection_in_description_never_corrupts_other_fields(): """Injection text inside the line description must either fall back OR parse safely: the amount binds to the LAST ' for ' (greedy), the po_number comes from the subject, and the payload stays confined to description.""" from decimal import Decimal parsed, method, _t, _r = try_deterministic_parse( load_email("adversarial", "adv-prompt-injection") ) if parsed is None: assert method == "ai_fallback" return assert parsed["po_number"] == "2D-12503765" assert parsed["total_amount"] == Decimal("55206.00") assert parsed["line_items"][0]["amount"] == Decimal("55206.00") assert "Ignore all previous instructions" in parsed["line_items"][0]["description"] def test_v10_url_id_invariant_across_corpus(): """Pre-lock corpus sweep (spec 8.5): on every harvested new_po (including the multi-line ones), the orders/ URL id equals the po_number digit suffix -- the invariant gate rule V10 enforces.""" stems = [("new-po", s) for s in NEW_PO_STEMS] + [ ("ai-fallback", s) for s in AI_FALLBACK_STEMS if s.startswith("new-po-") ] for subdir, stem in stems: candidate = template_parser.extract_new_po(load_email(subdir, stem)) po = candidate["po_number"] url = candidate["view_order_url"] m = re.match(r"^https://supplier\.coupahost\.com/orders/(\d+)\b", url or "") assert m and m.group(1) == po.split("-", 1)[1], (stem, po, url)