mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-09-30 07:13:13 +00:00
High-recall detector fan-out (injection/authz/secrets-crypto/iac-iam/logic) + proof-or-kill verifier. Gate PASSES: 1 confirmed medium, 0 confirmed critical/high. Confirmed finding fixed; several unverified-but-cheap hardenings applied since the emitter ships dark and activation is weeks out. - CONFIRMED medium (confused deputy): the rotation Lambda's generated invoke permission for secretsmanager.amazonaws.com carried no SourceAccount/SourceArn, so any account's Secrets Manager could invoke the rotator. Patched the generated CfnPermission in place (a second permission would be additive, not restrictive) to pin account + this secret ARN. - delivery + replay: refuse to follow receiver 3xx redirects (no-redirect opener) so live X-SH-* auth headers can't be forwarded to a receiver-chosen Location and an http:// Location can't slip past the https guard. Fixed the "unfollowed 3xx" comment that was factually wrong. - delivery: classify 401/403 as retryable (invalidate key cache + retry in order) instead of parking -- transient auth failures (rotation outran the TTL cache, clock skew) are availability events, not contract bugs. - envelope: build_event now genuinely total (guarded eventID / ApproximateCreationDateTime subscripts) per its own never-raise contract. - handler: catch-all so an unexpected per-record error (e.g. SQS park failure) reports only that record instead of failing the whole batch (which would re-deliver every earlier success for 24h); per-invocation emit/skip batch summary so a systemic silent drop is queryable/alarmable. - rotator: narrow the AWSCURRENT-read except to ResourceNotFound/JSONDecode (transient SM/KMS errors re-raise so the overlap key isn't silently dropped); kid uniqueness checked against ALL retained kids with a random suffix on collision (never reissue a kid for a different secret). - contract: skeleton-upsert required on ANY unknown work_order_id (not just comment-before-create) + monotonicity guard (ignore older updated_at), so a parked created or an out-of-order replay can't corrupt receiver state. Unverified/refuted findings left as-is with rationale: the two "high" logic claims (whole-batch crash triggers, ordering violation) were refuted on reachability (real stream records carry required fields; persistence writes strings only; full-state idempotent upsert absorbs the ordering gap). Signed kid/version binding (AUTHZ-002) declined: coordinated contract change, not cheap, no exploit with one algorithm/key.
211 lines
7 KiB
Python
211 lines
7 KiB
Python
"""Batch-loop tests for the SHOC webhook emitter handler (plan Phase 5).
|
|
|
|
Pins the partial-batch ordering contract: on a retryable failure at record i
|
|
the loop STOPS -- record i's DynamoDB SequenceNumber is reported via
|
|
report_batch_item_failures (so the ESM retries from it, in order) and later
|
|
records are never attempted; earlier in-batch successes are not re-delivered.
|
|
Non-retryable rejections park the full envelope on the rejected queue and the
|
|
loop CONTINUES (a contract bug must never block the shard). Skip records
|
|
(REMOVE / echo guard) produce no delivery attempts at all.
|
|
|
|
delivery.deliver is monkeypatched on the handler's own sibling instance (the
|
|
handler imports it by bare name); envelope.build_event runs for real, so
|
|
these batches exercise the true record -> envelope -> deliver path. SQS is a
|
|
plain fake on the handler's public ``sqs`` module global.
|
|
"""
|
|
|
|
import json
|
|
|
|
import pytest
|
|
|
|
from tests.support import load_lambda_module
|
|
|
|
WO_STREAM_ARN = (
|
|
"arn:aws:dynamodb:us-east-1:011934824531:table/WorkOrders"
|
|
"/stream/2026-07-23T00:00:00.000"
|
|
)
|
|
|
|
REJECTED_QUEUE_URL = (
|
|
"https://sqs.us-east-1.amazonaws.com/011934824531/workorder-shoc-emitter-rejected"
|
|
)
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def handler_mod():
|
|
return load_lambda_module("wo", "shoc_emitter/handler")
|
|
|
|
|
|
class _FakeSQS:
|
|
def __init__(self):
|
|
self.sent = []
|
|
|
|
def send_message(self, QueueUrl, MessageBody): # noqa: N803 (boto3 kwargs)
|
|
self.sent.append({"QueueUrl": QueueUrl, "MessageBody": MessageBody})
|
|
|
|
|
|
def _wo_record(event_id, sequence_number, event_name="INSERT", **image_overrides):
|
|
image = {
|
|
"work_order_id": {"S": f"wo-{event_id}"},
|
|
"wo_status": {"S": "new"},
|
|
"record_type": {"S": "new_work_order"},
|
|
}
|
|
image.update(image_overrides)
|
|
stream = {
|
|
"ApproximateCreationDateTime": 1784642602.0,
|
|
"SequenceNumber": sequence_number,
|
|
"StreamViewType": "NEW_AND_OLD_IMAGES",
|
|
}
|
|
if event_name != "REMOVE":
|
|
stream["NewImage"] = image
|
|
return {
|
|
"eventID": event_id,
|
|
"eventName": event_name,
|
|
"eventSource": "aws:dynamodb",
|
|
"eventSourceARN": WO_STREAM_ARN,
|
|
"dynamodb": stream,
|
|
}
|
|
|
|
|
|
def _wire(monkeypatch, handler_mod, outcomes):
|
|
"""Patch delivery.deliver with a per-delivery_id outcome table.
|
|
|
|
``outcomes`` maps delivery_id (the record eventID) to either a
|
|
("delivered"|"rejected", status) tuple or the string "retryable" (raise).
|
|
Returns the ordered list of attempted delivery_ids and the fake SQS.
|
|
"""
|
|
attempted = []
|
|
|
|
def _fake_deliver(webhook_event):
|
|
attempted.append(webhook_event["delivery_id"])
|
|
outcome = outcomes[webhook_event["delivery_id"]]
|
|
if outcome == "retryable":
|
|
raise handler_mod.delivery.RetryableDeliveryError(
|
|
"receiver returned 503", status_code=503
|
|
)
|
|
return outcome
|
|
|
|
monkeypatch.setattr(handler_mod.delivery, "deliver", _fake_deliver)
|
|
fake_sqs = _FakeSQS()
|
|
monkeypatch.setattr(handler_mod, "sqs", fake_sqs)
|
|
monkeypatch.setattr(handler_mod, "REJECTED_QUEUE_URL", REJECTED_QUEUE_URL)
|
|
return attempted, fake_sqs
|
|
|
|
|
|
def test_retryable_middle_record_stops_batch_and_reports_its_sequence(
|
|
monkeypatch, handler_mod
|
|
):
|
|
event = {
|
|
"Records": [
|
|
_wo_record("evt-1", "101"),
|
|
_wo_record("evt-2", "102"),
|
|
_wo_record("evt-3", "103"),
|
|
]
|
|
}
|
|
attempted, fake_sqs = _wire(
|
|
monkeypatch,
|
|
handler_mod,
|
|
{"evt-1": ("delivered", 200), "evt-2": "retryable"},
|
|
)
|
|
|
|
result = handler_mod.handler(event, None)
|
|
|
|
# EXACTLY the failed record's SequenceNumber: the ESM retries from it in
|
|
# order, and evt-1 (already delivered) is not re-delivered.
|
|
assert result == {"batchItemFailures": [{"itemIdentifier": "102"}]}
|
|
# The loop stopped at the failure: record 3 was never attempted.
|
|
assert attempted == ["evt-1", "evt-2"]
|
|
assert fake_sqs.sent == []
|
|
|
|
|
|
def test_rejected_record_parks_to_sqs_and_processing_continues(
|
|
monkeypatch, handler_mod
|
|
):
|
|
event = {
|
|
"Records": [
|
|
_wo_record("evt-1", "201"),
|
|
_wo_record("evt-2", "202"),
|
|
_wo_record("evt-3", "203"),
|
|
]
|
|
}
|
|
attempted, fake_sqs = _wire(
|
|
monkeypatch,
|
|
handler_mod,
|
|
{
|
|
"evt-1": ("delivered", 200),
|
|
"evt-2": ("rejected", 422),
|
|
"evt-3": ("delivered", 204),
|
|
},
|
|
)
|
|
|
|
result = handler_mod.handler(event, None)
|
|
|
|
# A 4xx rejection must NOT block the shard: no batch item failures, and
|
|
# the records after the rejection were still attempted.
|
|
assert result == {"batchItemFailures": []}
|
|
assert attempted == ["evt-1", "evt-2", "evt-3"]
|
|
|
|
# The full envelope was parked for operator replay.
|
|
assert len(fake_sqs.sent) == 1
|
|
assert fake_sqs.sent[0]["QueueUrl"] == REJECTED_QUEUE_URL
|
|
parked = json.loads(fake_sqs.sent[0]["MessageBody"])
|
|
assert set(parked) == {"envelope", "response_status"}
|
|
assert parked["response_status"] == 422
|
|
assert parked["envelope"]["delivery_id"] == "evt-2"
|
|
assert parked["envelope"]["event_type"] == "work_order.created"
|
|
assert parked["envelope"]["data"]["work_order_id"] == "wo-evt-2"
|
|
|
|
|
|
def test_skip_records_produce_no_delivery_calls(monkeypatch, handler_mod):
|
|
event = {
|
|
"Records": [
|
|
_wo_record("evt-1", "301", event_name="REMOVE"),
|
|
_wo_record("evt-2", "302", write_origin={"S": "shoc-write-api"}),
|
|
]
|
|
}
|
|
attempted, fake_sqs = _wire(monkeypatch, handler_mod, {})
|
|
|
|
result = handler_mod.handler(event, None)
|
|
|
|
assert result == {"batchItemFailures": []}
|
|
assert attempted == []
|
|
assert fake_sqs.sent == []
|
|
|
|
|
|
def test_unexpected_error_reports_record_not_whole_batch(monkeypatch, handler_mod):
|
|
# An unexpected exception (here: SQS park failure on a 4xx rejection) must
|
|
# NOT escape the loop -- that would fail the invocation and make the ESM
|
|
# re-deliver every earlier success for 24h. The offending record is
|
|
# reported so the ESM retries from it in order; evt-1 is not re-delivered.
|
|
event = {
|
|
"Records": [
|
|
_wo_record("evt-1", "401"),
|
|
_wo_record("evt-2", "402"),
|
|
_wo_record("evt-3", "403"),
|
|
]
|
|
}
|
|
attempted, fake_sqs = _wire(
|
|
monkeypatch,
|
|
handler_mod,
|
|
{
|
|
"evt-1": ("delivered", 200),
|
|
"evt-2": ("rejected", 400),
|
|
"evt-3": ("delivered", 200),
|
|
},
|
|
)
|
|
|
|
def _boom(QueueUrl, MessageBody): # noqa: N803 (boto3 kwargs)
|
|
raise RuntimeError("sqs unavailable")
|
|
|
|
monkeypatch.setattr(fake_sqs, "send_message", _boom)
|
|
|
|
result = handler_mod.handler(event, None)
|
|
|
|
assert result == {"batchItemFailures": [{"itemIdentifier": "402"}]}
|
|
# Stopped at the failing record; evt-3 not attempted.
|
|
assert attempted == ["evt-1", "evt-2"]
|
|
|
|
|
|
def test_empty_batch_returns_no_failures(monkeypatch, handler_mod):
|
|
attempted, _ = _wire(monkeypatch, handler_mod, {})
|
|
assert handler_mod.handler({"Records": []}, None) == {"batchItemFailures": []}
|
|
assert attempted == []
|