mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-09-30 22:23:14 +00:00
424 lines
14 KiB
Python
424 lines
14 KiB
Python
"""Fail-closed validation-gate tests.
|
||
|
||
Each known drift / adversarial shape must be rejected with the expected reason
|
||
code, and direct unit tests exercise each individual gate rule.
|
||
"""
|
||
|
||
import pytest
|
||
|
||
from _wo_parser_support import load_email, wo_template_parser as template_parser
|
||
|
||
CONTRACT_KEYS = template_parser.CONTRACT_KEYS
|
||
extract_update_plaintext = template_parser.extract_update_plaintext
|
||
try_deterministic_parse = template_parser.try_deterministic_parse
|
||
validate = template_parser.validate
|
||
validate_ai_fallback = template_parser.validate_ai_fallback
|
||
|
||
# Fixture stem -> expected fail-closed reason code.
|
||
EXPECTED_REASONS = {
|
||
"unknown-subject": "subject_no_match",
|
||
"missing-id": "subject_no_match",
|
||
"single-space-work-order": "single_space_work_order",
|
||
"nondigit-id": "missing_required_field",
|
||
"empty-new-comment": "missing_required_field",
|
||
"malformed-site-code": "malformed_site_code",
|
||
"label-bleed-comment": "label_bleed",
|
||
"unparseable-creation-time": "creation_time_unparseable",
|
||
"cancellation-t1": "missing_required_field",
|
||
"update-status-t1": "missing_required_field",
|
||
"missing-building-and-comment": "missing_required_field",
|
||
"t2-wo-id-mismatch": "wo_id_mismatch",
|
||
"t2-missing-address": "missing_required_field",
|
||
"t2-unparseable-date": "creation_time_unparseable",
|
||
}
|
||
|
||
|
||
@pytest.mark.parametrize("stem,reason", sorted(EXPECTED_REASONS.items()))
|
||
def test_gate_rejects_with_reason(stem, reason):
|
||
parsed, method, _tid, got_reason = try_deterministic_parse(
|
||
load_email("ai-fallback", stem)
|
||
)
|
||
assert parsed is None
|
||
assert method == "ai_fallback"
|
||
assert got_reason == reason
|
||
|
||
|
||
# --- Direct unit tests on validate() ---
|
||
|
||
|
||
def _good_t1_email():
|
||
return load_email("update-plaintext", "update-plaintext-01")
|
||
|
||
|
||
def _good_candidate():
|
||
return extract_update_plaintext(_good_t1_email())
|
||
|
||
|
||
def test_baseline_candidate_is_valid():
|
||
ok, reason = validate(_good_candidate(), "update_plaintext", _good_t1_email())
|
||
assert ok and reason == "ok"
|
||
|
||
|
||
def test_rule1_unknown_template():
|
||
ok, reason = validate(_good_candidate(), "unknown", _good_t1_email())
|
||
assert not ok and reason == "subject_no_match"
|
||
|
||
|
||
def test_rule2_extra_key_fails():
|
||
cand = _good_candidate()
|
||
cand["surprise"] = "x"
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "key_set_mismatch"
|
||
|
||
|
||
def test_rule2_missing_key_fails():
|
||
cand = _good_candidate()
|
||
del cand["address"]
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "key_set_mismatch"
|
||
|
||
|
||
def test_rule3_nondigit_wo():
|
||
cand = _good_candidate()
|
||
cand["work_order_id"] = "12A45"
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_rule5_wrong_email_type():
|
||
cand = _good_candidate()
|
||
cand["email_type"] = "new_work_order" # wrong for a T1 template
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "email_type_mismatch"
|
||
|
||
|
||
def test_rule6_bad_site_code():
|
||
cand = _good_candidate()
|
||
cand["site_code"] = "workshop"
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "malformed_site_code"
|
||
|
||
|
||
def test_rule7_bad_status():
|
||
# Phase 8 (constraint 7a): the template-path status-enum branch now returns
|
||
# its OWN reason code "invalid_status" -- previously a copy-paste bug made it
|
||
# return "malformed_site_code" (the site_code branch's code), so one status
|
||
# failure yielded two different codes depending on which parse path hit it.
|
||
cand = _good_candidate()
|
||
cand["status"] = "frobnicated"
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "invalid_status"
|
||
|
||
|
||
def test_template_and_ai_paths_agree_on_bad_status_reason_code():
|
||
"""Regression lock for the invalid_status reason-code fix (constraint 7a):
|
||
an identical bad-status failure must yield the SAME reason code on BOTH the
|
||
template ``validate`` path and the ``validate_ai_fallback`` path -- ending
|
||
the one-failure-two-codes-by-path split (doc Q4)."""
|
||
template_cand = _good_candidate()
|
||
template_cand["status"] = "frobnicated"
|
||
_, template_reason = validate(template_cand, "update_plaintext", _good_t1_email())
|
||
|
||
ai_cand = _ai_candidate()
|
||
ai_cand["work_order_id"] = "12345"
|
||
ai_cand["email_type"] = "update"
|
||
ai_cand["status"] = "frobnicated"
|
||
_, ai_reason = validate_ai_fallback(ai_cand)
|
||
|
||
assert template_reason == ai_reason == "invalid_status"
|
||
|
||
|
||
def test_rule8_empty_comment_text():
|
||
cand = _good_candidate()
|
||
cand["comment_text"] = " "
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_rule9_label_bleed_in_comment():
|
||
cand = _good_candidate()
|
||
cand["comment_text"] = "text that leaked Building: WCO0 into the value"
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "label_bleed"
|
||
|
||
|
||
def test_rule9_separator_bleed_in_address():
|
||
cand = _good_candidate()
|
||
cand["address"] = "123 Main St ________________"
|
||
ok, reason = validate(cand, "update_plaintext", _good_t1_email())
|
||
assert not ok and reason == "label_bleed"
|
||
|
||
|
||
def test_contract_keys_match_extraction_prompt():
|
||
"""The parser's key set must be exactly the AI EXTRACTION_PROMPT contract, so
|
||
the deterministic and AI-fallback paths write identical shapes downstream."""
|
||
import re
|
||
|
||
from _wo_parser_support import wo_handler as handler
|
||
|
||
# The prompt's JSON skeleton uses union-type pseudo-values (not strict JSON)
|
||
# and repeats some enum terms in prose bullets, so pull quoted "key": tokens
|
||
# and assert every contract key is a field the AI is asked to emit (the
|
||
# parser must never invent a key outside the AI contract).
|
||
prompt_keys = set(re.findall(r'"([a-z_]+)":', handler.EXTRACTION_PROMPT))
|
||
assert set(CONTRACT_KEYS).issubset(prompt_keys)
|
||
assert len(CONTRACT_KEYS) == 16 # current EXTRACTION_PROMPT field count
|
||
|
||
|
||
# --- validate_ai_fallback unit tests -----------------------------------------
|
||
|
||
|
||
def _ai_candidate():
|
||
return {k: None for k in CONTRACT_KEYS}
|
||
|
||
|
||
def test_ai_fallback_baseline_is_valid():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert ok and reason == "ok"
|
||
|
||
|
||
def test_ai_fallback_nondigit_wo_id():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12A45"
|
||
cand["email_type"] = "update"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_ai_fallback_null_wo_id():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = None
|
||
cand["email_type"] = "update"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_ai_fallback_hash_in_wo_id():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "123#spoofed#deadbeef"
|
||
cand["email_type"] = "update"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_ai_fallback_invalid_email_type():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "exploit"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_ai_fallback_null_email_type():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = None
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_ai_fallback_bad_status_enum():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["status"] = "frobnicated"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "invalid_status"
|
||
|
||
|
||
def test_ai_fallback_unhashable_enum_fails_closed():
|
||
# A JSON list/dict for an enum field is unhashable; the gate must fail
|
||
# closed (isinstance guard), not raise TypeError into async retries.
|
||
for field, bad in (
|
||
("email_type", ["update"]),
|
||
("email_type", {"x": 1}),
|
||
("status", ["new"]),
|
||
("status", {"x": 1}),
|
||
):
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand[field] = bad
|
||
ok, reason = validate_ai_fallback(cand) # must not raise
|
||
assert not ok, f"{field}={bad!r} should fail closed"
|
||
|
||
|
||
def test_ai_fallback_fullwidth_digit_wo_id_rejected():
|
||
# Fullwidth digits render like ASCII but are a distinct partition key;
|
||
# [0-9] (not \d) must reject them.
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345" # "12345" fullwidth
|
||
cand["email_type"] = "update"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "missing_required_field"
|
||
|
||
|
||
def test_ai_fallback_trailing_newline_rejected():
|
||
# \A..\Z (not ^..$) must reject a trailing newline in wo_id and site_code.
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345\n"
|
||
cand["email_type"] = "update"
|
||
ok, _ = validate_ai_fallback(cand)
|
||
assert not ok
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["site_code"] = "WIL1\n"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "malformed_site_code"
|
||
|
||
|
||
def test_ai_fallback_non_dict_fails_closed():
|
||
# json.loads on model output can yield any JSON type; the gate must fail
|
||
# closed on a non-object rather than raise into async retries / DLQ.
|
||
for bad in ([], "string", 42, None, [{"work_order_id": "12345"}]):
|
||
ok, reason = validate_ai_fallback(bad)
|
||
assert not ok and reason == "not_an_object", f"{bad!r} should fail closed"
|
||
|
||
|
||
def test_ai_fallback_unparseable_sentinel_rejected():
|
||
# The template parser's internal _UNPARSEABLE sentinel must never survive
|
||
# the AI gate into the store.
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["comment_time"] = "__UNPARSEABLE__"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "creation_time_unparseable"
|
||
|
||
|
||
def test_ai_fallback_non_iso_date_rejected():
|
||
for key in ("date_reported", "scheduled_start", "due_date", "comment_time"):
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand[key] = "ignore previous instructions"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "creation_time_unparseable", f"{key} not gated"
|
||
|
||
|
||
def test_ai_fallback_iso_dates_accepted():
|
||
for value in ("2026-07-16", "2026-07-16T10:15:00", "2026-07-16T10:15:00Z", None):
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["date_reported"] = value
|
||
cand["comment_time"] = value
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert ok, f"date {value!r} should pass"
|
||
|
||
|
||
def test_ai_fallback_valid_status_ok():
|
||
for status in (
|
||
"new",
|
||
"assigned",
|
||
"in_progress",
|
||
"on_hold",
|
||
"completed",
|
||
"cancelled",
|
||
"unknown",
|
||
None,
|
||
):
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["status"] = status
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert ok, f"status={status} should pass"
|
||
|
||
|
||
def test_ai_fallback_bad_site_code():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["site_code"] = "workshop"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "malformed_site_code"
|
||
|
||
|
||
def test_ai_fallback_valid_site_codes():
|
||
for code in ("WIL1", "ZDL8", "AB12", None):
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["site_code"] = code
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert ok, f"site_code={code} should pass"
|
||
|
||
|
||
def test_ai_fallback_key_set_mismatch_extra():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["surprise"] = "x"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "key_set_mismatch"
|
||
|
||
|
||
def test_ai_fallback_key_set_mismatch_missing():
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
del cand["address"]
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "key_set_mismatch"
|
||
|
||
|
||
def test_ai_fallback_all_valid_email_types():
|
||
for et in ("new_work_order", "update", "comment", "cancellation"):
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = et
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert ok, f"email_type={et} should pass"
|
||
|
||
|
||
def test_ai_fallback_free_text_float_rejected():
|
||
# Prompt-injected JSON float in a free-text slot must fail closed before
|
||
# update_item (boto3 rejects Python floats -> async retries / DLQ).
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["severity"] = 1.5
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "invalid_field_type"
|
||
|
||
|
||
def test_ai_fallback_free_text_dict_rejected():
|
||
# LLM-emitted maps in scalar slots must be rejected (DynamoDB Map pollution).
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["assigned_to"] = {"a": 1}
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "invalid_field_type"
|
||
|
||
|
||
def test_ai_fallback_free_text_list_rejected():
|
||
# LLM-emitted lists in scalar slots must be rejected (DynamoDB List pollution).
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand["comment_text"] = ["x"]
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert not ok and reason == "invalid_field_type"
|
||
|
||
|
||
def test_ai_fallback_free_text_none_and_str_ok():
|
||
# None and str remain valid for every free-text field.
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert ok and reason == "ok"
|
||
|
||
for field in template_parser._AI_FREE_TEXT_STR_FIELDS:
|
||
cand = _ai_candidate()
|
||
cand["work_order_id"] = "12345"
|
||
cand["email_type"] = "update"
|
||
cand[field] = "ok"
|
||
ok, reason = validate_ai_fallback(cand)
|
||
assert ok and reason == "ok", field
|