chore: tidy .claude/workflows — drop completed refactor phase scripts, keep remaining phase 6 (#120)

* chore: cleanup completed refactor workflow files

* chore: add remaining refactor phase 6 workflow
This commit is contained in:
Adam Moussa 2026-07-20 17:02:06 -04:00 • committed by GitHub
parent 65feb3a4d9
commit 004474fc1d
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 585 additions and 4338 deletions

View file

@ -1,428 +0,0 @@
export const meta = {
name: 'phase-0-deploy-guards',
description: 'Phase 0 of the procurement-ingest refactor (docs/refactor-evaluation.md): healthcheck early-return in both email processors, synchronous post-deploy smoke script wired into cd-cdk, PO cp-allowlist replaced with a non-recursive glob, and an AST bundle-consistency test. Built and adversarially verified on a branch off main, committed locally, never pushed (push is gated on /sh-security-review in the main loop).',
phases: [
{ title: 'Setup', detail: 'clean-tree check, branch feature/phase-0-deploy-guards off up-to-date main', model: 'haiku' },
{ title: 'Recon', detail: '4 parallel read-only mappers over handlers, CDK/alarms/CI, and test infra', model: 'haiku' },
{ title: 'Implement', detail: 'handlers on opus; smoke script, CDK glob + AST test on sonnet — disjoint file ownership, one branch', model: 'opus' },
{ title: 'Verify', detail: 'mechanical gates (pytest/ruff/synth/scope) on sonnet + 3 adversarial fable lenses' },
{ title: 'Fix', detail: 'opus fixer applies confirmed findings, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'README update, GPT-4.1 cross-family review of the diff, single commit via -F (no push)', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-0-deploy-guards'
// Pinned contract (memory lesson: define the shared contract BEFORE the
// parallel fan-out — parallel leaves can't see each other's choices).
const CONTRACT = `
HEALTHCHECK CONTRACT — pinned, do not deviate or "improve":
- Trigger: top-level direct-invoke payload only. In lambda handler ("handler"
in handler.py), the VERY FIRST statements: if the event is a dict and
event.get("healthcheck") is True -> return {"healthcheck": "ok"} immediately.
- Placement: BEFORE any boto3/S3 use, BEFORE ses_auth, BEFORE iterating
event["Records"]. It must not create any accept path for mail: real mail
events are S3 ObjectCreated events whose top-level keys AWS controls
("Records"); email content can never set a top-level event key.
- Telemetry: the healthcheck branch emits NO EMF metrics and NO log line
whose text could match the sender-auth-rejected metric-filter pattern
(read the filter pattern in cdk/*_stack.py before choosing any log text;
safest is a single log line exactly "healthcheck ok" or no logging).
That alarm pages at >=1 match, so two deploys in ~30 min must not page.
- SMOKE SCRIPT CONTRACT: scripts/post-deploy-smoke.sh invokes BOTH functions
(po-email-processor, workorder-email-processor) with payload
'{"healthcheck": true}' using: aws lambda invoke --invocation-type
RequestResponse. It must check the FunctionError field of the response
(an init ImportError returns HTTP 200 + FunctionError=Unhandled — exit-code
checks false-pass) AND that the payload equals {"healthcheck": "ok"}.
Non-zero exit on any failure; set -euo pipefail; region us-east-1.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 0 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push, do NOT run cdk deploy or
touch AWS resources; read-only aws CLI calls are also unnecessary).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 0".
Work ONLY in the files you are told you own. Other agents are concurrently
editing other files in this same working tree — do not read-depend on or
modify their files.
${CONTRACT}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'testsAdded', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
testsAdded: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string', description: 'exact commands run + pass/fail' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string', description: 'per-gate results; verbatim failure output' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean', description: 'true only with concrete file:line evidence' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
const TEXT = {
type: 'object',
required: ['summary'],
properties: { summary: { type: 'string' }, verdict: { type: 'string' } },
}
// ------------------------------------------------------------------- setup
phase('Setup')
const setup = await agent(`
In ${REPO}: verify the working tree is clean apart from untracked .coverage
and docs/refactor-evaluation.md (if anything ELSE is dirty, STOP and report a
blocker — do not stash or discard anything). Then:
git fetch origin && git checkout main && git pull --ff-only
git checkout -b ${BRANCH}
Also run: gh pr list --state open --json number,title,headRefName
(memory lesson: a branch cut fresh from main misses fixes sitting in unmerged
PRs — list them so the orchestrator can flag overlaps).
Return facts: current HEAD sha, branch created y/n, open PR list, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (setup && setup.blockers && setup.blockers.length) {
return { aborted: 'setup blockers', blockers: setup.blockers, openPRs: setup.facts }
}
log(`Branch ${BRANCH} ready. ${setup ? setup.summary : ''}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of lambdas/po/email_processor/handler.py and its tests dir.
Report: exact def line of the lambda handler; the first statements it executes
(S3 fetch? ses_auth call? Records iteration?) with line numbers; how existing
handler tests load the module and fake AWS (loader idiom, moto import-order
invariant in _po_parser_support.py); where a healthcheck test would naturally
live; any existing early-return branches. 10-20 precise facts.`,
{ label: 'recon:po-handler', model: 'haiku', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of lambdas/wo/email_processor/handler.py and its tests dir.
Same report shape: handler def line, first statements executed with line
numbers, test loader idiom (_wo_parser_support.py), where a healthcheck test
lives, existing early returns. 10-20 precise facts.`,
{ label: 'recon:wo-handler', model: 'haiku', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of cdk/po_stack.py, cdk/wo_stack.py and .github/workflows/.
Report: (1) the exact PO bundling command incl. the cp allowlist at
po_stack.py:~245-256 and the full file list of lambdas/po/email_processor/*.py
so the glob replacement can be proven identical-output; (2) the exact
sender-auth-rejected metric FILTER PATTERNS in both stacks (quote them
verbatim) and their alarm eval windows; (3) deploy.yaml's inputs to the
cd-cdk reusable workflow — fetch the pinned workflow file with
'gh api repos/Sea-Haven-Industries/.github/contents/.github/workflows/cd-cdk.yaml?ref=fd60e4c9041784f666ac0fdefb9bec3c7fbf5143'
(base64 -d the content) and confirm whether a post-deploy-script input exists
and its semantics (cwd, when it runs, failure handling). If it does NOT
exist, report that as a blocker with the closest available mechanism.
(4) WO bundling command for comparison. 15-25 precise facts.`,
{ label: 'recon:cdk-ci', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of test infrastructure: pytest.ini, tests/ (repo root),
tests/conftest.py, how CI (.github/workflows/ci.yaml) invokes pytest/ruff.
Report: how a NEW repo-root test file (tests/test_bundle_consistency.py)
would be collected; what it can import; whether tests/ has helpers for
locating lambda dirs; the ruff invocation used in CI. 8-15 precise facts.`,
{ label: 'recon:tests-ci', model: 'haiku', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n')
log(`Recon complete: ${reconOk.length}/4 mappers, ${reconBlockers.length} blockers`)
// --------------------------------------------------------------- implement
phase('Implement')
const implTasks = [
{ label: 'impl:po-healthcheck', model: 'opus', prompt: `${PREAMBLE}
YOU OWN: lambdas/po/email_processor/handler.py and NEW test file(s) under
lambdas/po/email_processor/tests/ ONLY.
Task: add the healthcheck early-return branch to the PO handler exactly per
the pinned contract. Add tests: (1) {"healthcheck": true} returns
{"healthcheck": "ok"} with ZERO S3 calls, ZERO ses_auth calls, ZERO DynamoDB
writes, ZERO metric emission (assert via monkeypatch/fakes per the existing
loader idiom — preserve the moto-before-handler import ordering); (2) a normal
S3 mail event is completely unaffected (an existing golden-path test still
passing is necessary but ALSO assert a mail event containing the string
"healthcheck" in its email body does NOT take the branch).
Run before returning: ruff check lambdas/po, ruff format lambdas/po --check,
pytest lambdas/po/email_processor/tests -q --no-cov.
Recon context:\n${pack}` },
{ label: 'impl:wo-healthcheck', model: 'opus', prompt: `${PREAMBLE}
YOU OWN: lambdas/wo/email_processor/handler.py and NEW test file(s) under
lambdas/wo/email_processor/tests/ ONLY.
Task: identical healthcheck branch + tests for the WO handler, per contract,
mirroring the PO task one-for-one but using the WO test loader idiom.
Run before returning: ruff check lambdas/wo, ruff format lambdas/wo --check,
pytest lambdas/wo/email_processor/tests -q --no-cov.
Recon context:\n${pack}` },
{ label: 'impl:smoke-script', model: 'sonnet', prompt: `${PREAMBLE}
YOU OWN: scripts/post-deploy-smoke.sh (new) and .github/workflows/deploy.yaml ONLY.
Task: write the synchronous smoke script per the pinned SMOKE SCRIPT CONTRACT
(both functions, RequestResponse, FunctionError field check + payload check,
set -euo pipefail, executable bit, shellcheck-clean). Wire it into deploy.yaml
via the cd-cdk reusable workflow's post-deploy-script input per the recon
facts below — if recon reported that input missing, implement the closest
mechanism recon identified and record a blocker note instead of inventing
workflow inputs. Do NOT restructure deploy.yaml otherwise.
Run before returning: bash -n scripts/post-deploy-smoke.sh, and
python3 -c "import yaml,sys;yaml.safe_load(open('.github/workflows/deploy.yaml'))".
Recon context:\n${pack}` },
{ label: 'impl:bundle-glob-ast', model: 'sonnet', prompt: `${PREAMBLE}
YOU OWN: cdk/po_stack.py (ONLY the email-processor bundling command block,
~lines 241-258) and tests/test_bundle_consistency.py (new) ONLY.
Task A: replace the four-file cp allowlist with a non-recursive glob:
"cp ./*.py /asset-output/". Keep the pip install line untouched. Update the
warning comment to explain the glob + that the AST test now enforces
consistency. PROVE identical output: list lambdas/po/email_processor/*.py
(non-recursive) and confirm the set equals {handler, ses_auth,
template_parser, derived_fields}.py plus any other top-level .py that SHOULD
ship; if extras exist (e.g. prompts or __init__), state explicitly whether
shipping them is a no-op and why. tests/ and package/ are directories, so a
non-recursive ./*.py glob never matches them.
Task B: write tests/test_bundle_consistency.py: for EACH pipeline (po, wo),
ast-parse email_processor/handler.py, collect top-level "import X" /
"from X import ..." names, filter to first-party siblings (X.py exists in the
same dir), and assert the bundling guarantee ships them — for PO: assert the
po_stack.py bundling command contains the glob "cp ./*.py"; for WO: assert
its bundling command copies them (read wo_stack.py to see its current cp -r
form and assert accordingly). The test must FAIL if someone reverts the glob
to an allowlist missing a sibling — include a unit-level check that simulates
an allowlist command string missing derived_fields.py and asserts the
detection logic catches it. No AWS/boto3/synth in the test — pure
ast + file reads, fast.
Run before returning: ruff check cdk tests, pytest tests/test_bundle_consistency.py -q --no-cov.
Recon context:\n${pack}` },
]
const impl = await parallel(implTasks.map(t => () =>
agent(t.prompt, { label: t.label, model: t.model, phase: 'Implement', schema: IMPL })))
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/4 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'lambdas/po/email_processor/handler.py',
'lambdas/po/email_processor/tests/',
'lambdas/wo/email_processor/handler.py',
'lambdas/wo/email_processor/tests/',
'scripts/post-deploy-smoke.sh',
'.github/workflows/deploy.yaml',
'cdk/po_stack.py',
'tests/test_bundle_consistency.py',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification (trust-but-verify — do not rely on implementers'
self-reports). Run ALL of, reporting each verbatim on failure:
1. pytest -q --no-cov (repo root — all 3 roots, expect ~568+new all green)
2. ruff check .
3. ruff format --check .
4. npx cdk synth po-ingest -q && npx cdk synth workorder-ingest -q
(run inside cdk/; selectors are ARTIFACT IDs, not stack_name)
5. bash -n scripts/post-deploy-smoke.sh; test -x scripts/post-deploy-smoke.sh
6. git status --porcelain — every modified/added path must fall under:
${EXPECTED_SCOPE.join(', ')} (plus untracked .coverage,
docs/refactor-evaluation.md, .claude/workflows/). List violations.
7. Grep the healthcheck branch in both handlers: confirm it precedes any S3
get_object and any ses_auth call by line number.
YOU MAY NOT edit any file. passed=true only if every gate is green and scope
is clean.`
const lenses = [
{ key: 'security', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — security lens. Try to REFUTE the safety of this diff
(git diff main). Attack: (1) can any S3/SES-delivered email reach the
healthcheck branch or any other new early-return (top-level key forgery via
event shape, weird typing like event={"healthcheck":"true"} vs True)?
(2) does placement before ses_auth weaken fail-closed behavior in ANY path?
(3) does the smoke script or deploy.yaml change introduce injection (unquoted
vars, payload echoed into shell)? (4) does the AST test import or execute
handler code (it must not)? confirmed=true ONLY with a concrete exploit
sketch + file:line.` },
{ key: 'telemetry', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — telemetry/alarm lens. Read the sender-auth-rejected
metric FILTER PATTERNS in cdk/po_stack.py and cdk/wo_stack.py verbatim, then
try to prove a healthcheck invocation (incl. Lambda platform START/REPORT
lines and any new log text) produces a filter match — that alarm pages at
>=1. Also verify: zero EMF emission on the healthcheck path; ParseMethod
metric contract untouched (PO emits ai_fallback BEFORE the Bedrock call —
must not have moved); no alarm/math changes snuck into po_stack.py beyond
the bundling block. confirmed=true only with file:line evidence.` },
{ key: 'bundling', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — bundling/CI lens. (1) Enumerate
lambdas/po/email_processor/*.py and prove the new glob ships EXACTLY the
right set vs the old allowlist {handler,ses_auth,template_parser,
derived_fields}.py — flag any extra top-level .py whose shipping is NOT a
proven no-op, and confirm tests/ + package/ cannot match a non-recursive
./*.py. (2) Mutation-test the AST consistency test: temporarily copy the
detection logic and feed it an allowlist string missing derived_fields.py —
does it fail? (do this in /tmp scratch, not by editing repo files).
(3) Smoke script: does it actually catch FunctionError on HTTP 200 (trace the
aws CLI output handling — --query vs jq parsing), and does a missing function
name or region default break it? (4) deploy.yaml: is the post-deploy wiring
consistent with the cd-cdk reusable workflow's actual input names (re-fetch
the pinned file via gh api if needed)? confirmed=true only with evidence.` },
]
let round = 0
let checks = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
confirmed = results.slice(1).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && (f.severity === 'critical' || f.severity === 'high' || f.severity === 'medium'))
const mechanicalGreen = checks && checks.passed
log(`Verify round ${round}: mechanical ${mechanicalGreen ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (mechanicalGreen && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit any Phase-0-owned file listed here:
${EXPECTED_SCOPE.join(', ')}.
Fix EVERY item below with the minimal change; do not expand scope; keep the
pinned contract intact. Re-run the specific failing check/test for each fix.
MECHANICAL FAILURES:\n${checks ? checks.details : '(mechanical agent died — rerun everything)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
openPRs: setup ? setup.facts : [],
}
}
// ----------------------------------------------------------------- package
phase('Package')
const readme = await agent(`${PREAMBLE}
YOU OWN: README.md only. Document (in the style of the existing README):
the {"healthcheck": true} direct-invoke contract on both processors, the
post-deploy smoke gate (what it checks, that FunctionError is the signal),
and the PO bundling glob + AST consistency test (replacing the allowlist
note if one exists). Same-commit README updates are a handbook requirement.
Run: ruff format --check . still clean (README is md, but confirm no stray
edits). Return filesChanged.`,
{ label: 'package:readme', model: 'sonnet', phase: 'Package', schema: IMPL })
const crossReview = await agent(`${PREAMBLE}
The healthcheck branch adds a new event-contract field to both Lambda
handlers — the handbook mandates a cross-family review for handler-signature
/ event-shape changes. Run exactly:
cd ${REPO} && git diff main > /tmp/phase0.diff
python3 ~/Documents/repositories/seahaven/security-review/cross_review.py \
"Review this diff for breaking changes. Context: procurement-ingest refactor Phase 0 per docs/refactor-evaluation.md — healthcheck early-return added to both email-processor handlers (new top-level event field, direct-invoke only), post-deploy smoke script, PO bundling cp-allowlist replaced with ./*.py glob, AST bundle-consistency test. Diff follows: $(cat /tmp/phase0.diff)"
(cross_review.py is stateless/one-shot — the diff must be inline.) Return its
verdict VERBATIM in summary, and set verdict to one of: PASS / FIX / BLOCK
based on its highest finding. Do not fix anything yourself.`,
{ label: 'package:cross-review', model: 'sonnet', phase: 'Package', schema: TEXT })
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent. Steps:
1. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md
and follow it exactly.
2. git add: the Phase-0 files (${EXPECTED_SCOPE.join(', ')}), README.md, AND
docs/refactor-evaluation.md (untracked evaluation report — must land with
this first refactor PR or it is lost) AND .claude/workflows/phase-0-deploy-guards.js.
Do NOT add .coverage. Verify with git status that nothing unexpected is staged.
3. ONE commit. Write the message to /tmp/phase0-commit-msg.txt and use
git commit -F /tmp/phase0-commit-msg.txt (backticks in -m get eaten by zsh).
Suggested subject: "feat: deploy-pipeline guards — healthcheck, smoke gate, bundle glob + AST test (refactor phase 0)".
NO AI attribution / Co-Authored-By lines.
4. Do NOT push. Return the commit sha + shortstat in summary.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
commit: commit ? commit.summary : 'commit agent died — commit manually',
crossFamilyReview: crossReview ? { verdict: crossReview.verdict, detail: crossReview.summary } : 'NOT RUN — outstanding',
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged).concat(readme ? readme.filesChanged : []),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
openPRsAtBranchTime: setup ? setup.facts : [],
outstandingGates: [
'/sh-security-review (MANDATORY before push — handler = untrusted-input surface); run in the main loop on the committed diff',
'if cross-family verdict is FIX/BLOCK: resolve, re-run cross_review.py on the amended diff',
'push + PR + gh pr checks green',
'deploy-then-merge: deploy from branch, smoke green, one real PO + WO email each showing ParseMethod=template, all alarms green, THEN merge',
],
}

View file

@ -1,482 +0,0 @@
export const meta = {
name: 'phase-2-bundling-root',
description: 'Phase 2 of the procurement-ingest refactor (docs/refactor-evaluation.md): widen both email-processor asset roots to ../lambdas (making lambdas/shared/ reachable for Phase 3) with scoped cp globs, add exclude lists (from_asset ignores .gitignore — the stale 44 MB package/ dir would poison asset hashes), and add __pycache__ excludes to the three plain from_asset calls. ZERO code change under lambdas/ — the workflow proves bundle parity by diffing the locally-synthed staged asset against the currently-deployed zip. Committed locally, never pushed.',
phases: [
{ title: 'Setup', detail: 'verify Phases 0+1 merged into base, branch feature/phase-2-bundling-root', model: 'haiku' },
{ title: 'Recon', detail: '3 mappers: all five from_asset sites, deployed zip file lists (read-only AWS), AST-test shapes' },
{ title: 'Spec', detail: 'serial opus spec: exact new bundling commands, exclude lists, and the parity-comparison rules', model: 'opus' },
{ title: 'Implement', detail: 'sonnet per stack file + sonnet for AST-test/README updates — disjoint files', model: 'sonnet' },
{ title: 'Verify', detail: 'mechanical gates + staged-vs-deployed parity + determinism + 2 fable lenses' },
{ title: 'Fix', detail: 'opus fixer, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'single commit via -F (no push)', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-2-bundling-root'
const BASE = (args && args.base) || 'main'
const CONSTRAINTS = `
PINNED BEHAVIORAL CONSTRAINTS (docs/refactor-evaluation.md Phase 2 — violating any is a build failure):
1. ZERO diff under lambdas/: git diff ${BASE}...HEAD -- lambdas/ must be
EMPTY. This phase moves NO code — only cdk/, tests/test_bundle_consistency.py,
README.md (and docs/ if needed) may change.
2. Both email processors: Code.from_asset("../lambdas") with the bundling
command rewritten for the new cwd (bundling cwd IS the asset root):
pip install ... -r po/email_processor/requirements.txt -t /asset-output &&
cp po/email_processor/*.py /asset-output/ (resp. wo/email_processor).
The -r path change is REQUIRED. PRESERVE the pip
"--platform manylinux2014_aarch64 --only-binary=:all:" pin exactly — its
removal shipped x86 wheels into the ARM64 function and caused a 100%
outage (PR #34); it is non-negotiable.
3. Both widened from_asset calls get
exclude=['**/__pycache__/**', '**/tests/**', '**/package/**'].
from_asset does NOT honor .gitignore; without the package/ exclude the
stale untracked 44 MB lambdas/po/email_processor/package/ dir diverges
LOCAL vs CI asset hashes (CI never has it) and forces spurious redeploys.
4. WO prod-zip shrinkage is DELIBERATE cleanup, not an accident to fix: the
current 'cp -r .' ships tests/ (real scrubbed .eml fixtures), __pycache__,
and requirements.txt in the production zip. The acceptance criterion for
WO is "runtime-imported module set unchanged + smoke", NOT byte-identical
zip. Byte-identical first-party file set applies to PO ONLY. Document the
shrinkage explicitly (list every file class that drops out).
5. The three plain from_asset calls (po web_ui, po site_extractor, wo web_ui
— locate them, line numbers have shifted since the doc) each gain
exclude=['**/__pycache__/**'] so local __pycache__ stops making their
asset hashes nondeterministic. Nothing else about them changes.
6. tests/test_bundle_consistency.py must be updated IN THE SAME CHANGE — it
deliberately change-detects both bundling commands and will fail red on
the new shapes. Update it to recognize the scoped glob
'cp po/email_processor/*.py' (resp. wo) as the new unconditionally-safe
shape, WITHOUT loosening it: the allowlist-revert detection, the
detection-logic mutation test, and the PO_EXPECTED_TOP_LEVEL_MODULES
exact-set pin must all keep their teeth. A bare-substring match that any
commented-out glob satisfies is a regression.
7. cdk diff on BOTH stacks must show ONLY the two functions' Code/asset
property changes (+ CDK metadata). No logical-ID changes, no alarm, IAM,
table, env, or function-config deltas of any kind.
8. Do NOT delete lambdas/po/email_processor/package/ (untracked, local-only;
deletion is Adam's call — with the exclude in place it is hash-neutral).
Do NOT touch requirements.txt contents (Phase 7 scope).
9. AWS access is READ-ONLY (get-function, downloading the deployed zip via
its presigned URL). NEVER cdk deploy, never invoke, never mutate.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 2 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 2".
Work ONLY in the files you are told you own; other agents are concurrently
editing other files in this same working tree.
${CONSTRAINTS}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const SPEC = {
type: 'object',
required: ['poBundling', 'woBundling', 'plainAssetEdits', 'astTestChanges', 'parityRules', 'notes'],
properties: {
poBundling: { type: 'string', description: 'the complete new po_stack.py from_asset block: asset path, full command string, exclude list — exact code' },
woBundling: { type: 'string', description: 'same for wo_stack.py' },
plainAssetEdits: { type: 'string', description: 'the three plain from_asset calls: current file:line + exact exclude addition each' },
astTestChanges: { type: 'string', description: 'exactly how test_bundle_consistency.py recognizes the new scoped-glob shapes without losing revert detection; which assertions/regexes change and to what' },
parityRules: { type: 'string', description: 'the staged-asset vs deployed-zip comparison rules: strict first-party .py set equality for PO; WO expected-shrinkage list + runtime-module retention; how dependency version drift is tolerated' },
notes: { type: 'string' },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const PARITY = {
type: 'object',
required: ['passed', 'poVerdict', 'woVerdict', 'details'],
properties: {
passed: { type: 'boolean' },
poVerdict: { type: 'string', description: 'PO staged vs deployed: identical first-party set? full evidence' },
woVerdict: { type: 'string', description: 'WO: runtime modules retained + exact shrinkage list' },
determinism: { type: 'string', description: 'repeat-synth + injected-__pycache__ hash results' },
details: { type: 'string' },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
// ------------------------------------------------------------------- setup
phase('Setup')
const setup = await agent(`
In ${REPO}:
1. SEQUENCING GATE — Phases 0 AND 1 must be merged into ${BASE} (Phase 1
also edits cdk/po_stack.py; building Phase 2 against a pre-Phase-1 stack
file guarantees a conflict). git fetch origin, then verify on
origin/${BASE}: (a) the healthcheck branch exists in both handlers and
tests/test_bundle_consistency.py exists (Phase 0); (b) validate_ai_fallback
exists in the PO pipeline and po_stack.py contains the
ai-fallback-rejected alarm (Phase 1). If either is missing, STOP with a
blocker naming the unmerged phase and do nothing else.
2. Verify clean tree (untracked .coverage/.claude/ fine; other dirt = blocker,
never stash or discard).
3. git checkout ${BASE} && git pull --ff-only && git checkout -b ${BRANCH}
4. gh pr list --state open --json number,title,headRefName
Return: HEAD sha, gate evidence, open PRs, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (!setup || (setup.blockers && setup.blockers.length)) {
return { aborted: 'setup blockers', blockers: setup ? setup.blockers : ['setup agent died'], facts: setup ? setup.facts : [] }
}
log(`Branch ${BRANCH} ready off ${BASE}. ${setup.summary}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of ALL FIVE Code.from_asset sites in cdk/po_stack.py and
cdk/wo_stack.py: for each, quote verbatim with current file:line — the asset
path, full bundling command (if bundled) or plain form, and any existing
exclude. Also: both email_processor requirements.txt paths + contents (the
pip -r path must be rewritten); the exact set of top-level .py files in
lambdas/po/email_processor and lambdas/wo/email_processor (non-recursive);
every subdirectory of each (tests/, package/, __pycache__ presence); and
whether lambdas/ contains anything else a widened ../lambdas asset root
would stage (shared/ existing yet? stray files at lambdas/ top level?).
15-25 precise facts.`,
{ label: 'recon:asset-sites', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only AWS recon (region us-east-1, READ-ONLY): for po-email-processor
and workorder-email-processor run aws lambda get-function, download each
Code.Location presigned zip to a scratch dir, and produce the COMPLETE file
list of each deployed zip (unzip -l), separating (a) first-party top-level
.py files, (b) pip-installed dependency dirs (name + version from
.dist-info), (c) everything else (tests/, fixtures, __pycache__,
requirements.txt — expected in the WO zip today). Also record each
function's CodeSha256. This is the parity baseline the new bundles are
judged against. Return the categorized lists as facts.`,
{ label: 'recon:deployed-zips', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of tests/test_bundle_consistency.py: quote the exact
regexes/assertions that will change — the PO executed-glob pin, the WO
recursive-copy pin, _bundling_ships_all's two unconditionally-safe shapes,
_extract_bundling_command's exactly-one-command demand (both stacks still
have exactly one bundled function after Phase 2 — confirm), the
PO_EXPECTED_TOP_LEVEL_MODULES pin, and the mutation test. State which
assertions fail red against the Phase 2 command shapes (expected: PO glob
pin and WO cp -r pin both fail). 8-15 facts.`,
{ label: 'recon:ast-test', model: 'haiku', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n')
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
log(`Recon complete: ${reconOk.length}/3 mappers, ${reconBlockers.length} blockers`)
// -------------------------------------------------------------------- spec
phase('Spec')
const spec = await agent(`${PREAMBLE}
You are the SPEC agent — pin every exact string before parallel
implementation. Using the recon pack and your own reads, produce:
- poBundling / woBundling: the complete replacement from_asset blocks as
exact code — asset path "../lambdas", full bash -c command (pip line with
rewritten -r path and the preserved manylinux2014_aarch64 pin, then the
scoped non-recursive glob cp), exclude list per constraint 3. Mind the
bundling cwd semantics: the container mounts the ASSET ROOT (../lambdas)
as the working dir, so all paths are relative to lambdas/.
- plainAssetEdits: the three plain from_asset calls with exact exclude
additions.
- astTestChanges: precise edits to test_bundle_consistency.py — the new
scoped-glob regex (must match 'cp po/email_processor/*.py /asset-output/'
as executed, still comment-strip-proof, still reject narrowed or
commented-out variants), what replaces the WO cp -r pin, and confirmation
the mutation test + exact-set pin survive unchanged in spirit.
- parityRules: exactly how Verify judges the new bundles against the
deployed-zip baseline from recon: PO = first-party top-level .py set must
be IDENTICAL; dependency packages compared by NAME (version drift from the
floor-pinned boto3 is tolerated and noted, not failed); nothing new may
appear. WO = every first-party module in the CURRENT deployed zip that the
handler imports (cross-check the AST sibling list) must be retained;
expected drop list = tests/ + fixtures + __pycache__ + requirements.txt
and NOTHING else may silently vanish; nothing new may appear.
Recon pack:\n${pack}`,
{ label: 'spec:pin-strings', model: 'opus', phase: 'Spec', schema: SPEC })
if (!spec) return { aborted: 'spec agent died — rerun workflow', reconBlockers }
const specBlock = `BINDING SPEC (implement EXACTLY this):\n${JSON.stringify(spec, null, 2)}`
log('Spec pinned: bundling commands, excludes, AST-test edits, parity rules')
// --------------------------------------------------------------- implement
phase('Implement')
const impl = await parallel([
() => agent(`${PREAMBLE}
YOU OWN: cdk/po_stack.py ONLY. Apply spec.poBundling (email processor
from_asset: root, command, exclude) and the po web_ui + site_extractor
exclude additions from spec.plainAssetEdits. Touch NOTHING else in the file
(alarms, IAM, tables are all off-limits). Preserve/adapt the bundling
warning comment so it stays true.
Run before returning: ruff check cdk && cd cdk && npx cdk synth po-ingest -q
(this runs Docker bundling — confirm the staged asset in cdk.out contains
exactly the expected files and note the asset hash in your summary).
${specBlock}`,
{ label: 'impl:po-stack', model: 'sonnet', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: cdk/wo_stack.py ONLY. Apply spec.woBundling and the wo web_ui
exclude from spec.plainAssetEdits. Touch nothing else.
Run before returning: ruff check cdk && cd cdk && npx cdk synth
workorder-ingest -q (artifact-id selector, NOT stack_name) — confirm staged
asset contents + hash in your summary.
${specBlock}`,
{ label: 'impl:wo-stack', model: 'sonnet', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: tests/test_bundle_consistency.py and README.md ONLY.
Task A: apply spec.astTestChanges. The test must PASS against the new stack
files and still FAIL against: a reverted filename allowlist missing a
sibling, a commented-out glob, and a narrowed glob (add/adjust the mutation
tests to cover the new shapes — e.g. 'cp po/email_processor/handler.py'
alone must not satisfy the scoped-glob pin).
Task B: README — update the Deploy-Pipeline Guards bundling paragraph and
the CI/CD + repo-layout sections for the widened ../lambdas asset root;
document the deliberate WO prod-zip shrinkage (what dropped and why that is
cleanup, per constraint 4) and the exclude rationale (from_asset ignores
.gitignore; package/ hash poisoning).
Run before returning: pytest tests/test_bundle_consistency.py -q --no-cov
(must be green against the OTHER agents' stack edits — if they have not
landed yet, poll by re-running up to ~10 min before reporting a blocker),
ruff check tests.
${specBlock}`,
{ label: 'impl:ast-test-readme', model: 'sonnet', phase: 'Implement', schema: IMPL }),
])
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/3 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'cdk/po_stack.py',
'cdk/wo_stack.py',
'tests/test_bundle_consistency.py',
'README.md',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification — trust nothing self-reported. Run ALL gates,
quoting failures verbatim:
1. pytest -q --no-cov (repo root)
2. ruff check . && ruff format --check .
3. cd cdk && npx cdk synth po-ingest -q && npx cdk synth workorder-ingest -q
4. cd cdk && npx cdk diff po-ingest ; npx cdk diff workorder-ingest —
the ONLY resource deltas allowed are the two email-processor functions'
Code/S3Key (asset hash) changes + CDK metadata. Any alarm/IAM/env/config/
logical-ID delta = FAIL (constraint 7). Paste the diff summaries.
5. ZERO-CODE-MOVE invariant: git diff ${BASE}...HEAD -- lambdas/ must output
NOTHING.
6. git status scope: every change under ${EXPECTED_SCOPE.join(', ')} only
(untracked .coverage/.claude/ tolerated).
passed=true only if all green. YOU MAY NOT edit files.`
const parityPrompt = `${PREAMBLE}
You are the ARTIFACT-PARITY verifier — the load-bearing gate of this phase.
Everything is local synth + read-only AWS.
1. cd cdk && npx cdk synth po-ingest -q -o /tmp/phase2-synth && npx cdk
synth workorder-ingest -q -o /tmp/phase2-synth. Locate each email
processor's staged bundled asset dir in /tmp/phase2-synth (the asset.*
dir containing handler.py).
2. Re-download both deployed zips (aws lambda get-function Code.Location,
region us-east-1) fresh — do not trust a recon cache.
3. Judge per the spec parityRules: PO first-party top-level .py set staged
vs deployed IDENTICAL (list both sets); dependency packages by name with
version drift noted; NOTHING new. WO: every AST-derived handler sibling
retained; the drop list is EXACTLY tests//fixtures/__pycache__/
requirements.txt-class files — enumerate every dropped path class and
every added path; anything outside the expected classes = FAIL.
4. DETERMINISM: run the po-ingest synth twice into two fresh -o dirs —
asset hashes must be identical. Then create a throwaway
__pycache__/junk.pyc under lambdas/po/web_ui/ AND a dummy file under
lambdas/po/email_processor/package/ (create the dir if absent, remember
to DELETE everything you created afterwards), re-synth, and confirm NO
asset hash changed (proves the excludes + the package/ hash-poisoning
fix actually work). Report before/after hashes.
5. Sanity: the staged PO asset must NOT contain web_ui/site_extractor
sources, tests/, package/, or any .eml — the widened root stages more
context but the cp glob + excludes must keep the OUTPUT clean.
passed=true only if every check holds.`
const lenses = [
{ key: 'bundling-semantics', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — bundling-semantics lens. Read the full cdk diff
(git diff ${BASE}) plus aws-cdk-lib's documented from_asset/BundlingOptions
semantics and try to REFUTE correctness: (1) is the bundling container cwd
really the widened asset root, making 'pip install -r
po/email_processor/requirements.txt' and 'cp po/email_processor/*.py'
resolve correctly — or does an image workdir default break it? (2) do the
exclude patterns ('**/__pycache__/**' etc.) apply to ASSET STAGING (hash
input) and not merely the docker mount — i.e. does the package/ exclude
actually stop local-vs-CI hash divergence? (3) does exclude affect what the
bundling container can SEE, and could excluding tests/ break the pip
install or cp step? (4) non-recursive scoped glob: could bash glob
expansion inside the container (sh vs bash, nullglob) change behavior vs
the Phase 0 top-level glob? (5) does widening the asset root change the
ASSET HASH INPUTS such that unrelated wo/ file edits now redeploy the PO
function (and vice versa) — if so, state the blast radius plainly so it is
documented, not discovered. confirmed=true only with concrete evidence
(CDK docs/source or a reproduced synth).` },
{ key: 'guard-integrity', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — guard-integrity lens. The Phase 0 guards must come
through Phase 2 with their teeth intact. (1) Attack the updated
test_bundle_consistency.py: does the new scoped-glob regex accept a
commented-out glob, a narrowed single-file cp, a glob for the WRONG
pipeline dir (cp wo/email_processor/*.py in po_stack), or a cp missing
/asset-output? Does the WO assertion still reject a narrowed recursive
copy? Run the test against hand-mutated command strings to prove each
rejection (scratch copies, not repo edits). (2) Does
PO_EXPECTED_TOP_LEVEL_MODULES still bound what ships (the glob is now
scoped — is the pin still checking the right directory)? (3) The smoke
script + healthcheck are untouched by this diff — confirm zero diff to
scripts/ and lambdas/. (4) README claims about bundling must match the new
reality — no stale Phase 0 text contradicting the widened root.
confirmed=true only with file:line evidence.` },
]
let round = 0
let checks = null
let parity = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
() => agent(parityPrompt, { label: `verify:parity-r${round}`, model: 'opus', phase: 'Verify', schema: PARITY }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
parity = results[1]
confirmed = results.slice(2).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && f.severity !== 'low')
const green = checks && checks.passed && parity && parity.passed
log(`Verify round ${round}: mechanical ${checks && checks.passed ? 'GREEN' : 'RED'}, parity ${parity && parity.passed ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (green && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit files under: ${EXPECTED_SCOPE.join(', ')}.
Fix EVERY item minimally; the binding spec and 9 pinned constraints hold (a
finding that conflicts with a constraint is reported, not "fixed"). Re-run
the specific failing gate per fix.
MECHANICAL:\n${checks ? checks.details : '(agent died — rerun all)'}
PARITY:\n${parity ? parity.details : '(agent died — rerun all)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}
${specBlock}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && parity && parity.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
parity,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
}
}
// ----------------------------------------------------------------- package
phase('Package')
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent:
1. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md.
2. git add only: ${EXPECTED_SCOPE.join(', ')} and
.claude/workflows/phase-2-bundling-root.js. NOT .coverage. Verify staged
set with git status.
3. ONE commit via git commit -F /tmp/phase2-commit-msg.txt (write the message
file first; backticks in -m get eaten by zsh). Suggested subject:
"feat: widen email-processor asset roots to lambdas/ with scoped globs + excludes (refactor phase 2)"
Body: why (shared/ reachability for Phase 3), the WO shrinkage list, the
package/ hash-poisoning exclude, parity evidence one-liner.
NO AI attribution / Co-Authored-By lines.
4. Do NOT push. Return commit sha + shortstat.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
base: BASE,
commit: commit ? commit.summary : 'commit agent died — commit manually',
parityEvidence: parity ? { po: parity.poVerdict, wo: parity.woVerdict, determinism: parity.determinism } : null,
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
outstandingGates: [
'push + PR + gh pr checks green (no /sh-security-review required — no untrusted-input/auth/IAM surface; pre-push scanners still run; cross-family review not required — no IAM or handler-signature change)',
'deploy-then-merge: deploy from branch, then the LIVE gate — aws lambda get-function zip file-list diff (PO first-party set identical; WO shrinkage exactly as documented), smoke green, one real PO + WO email each with ParseMethod=template, alarms green, THEN merge',
'optional cleanup for Adam (hash-neutral once excludes are deployed): delete the untracked 44 MB lambdas/po/email_processor/package/ dir (doc Phase 7 item, endorsed to do with/after Phase 2)',
],
}

View file

@ -1,679 +0,0 @@
export const meta = {
name: 'phase-3-shared-extraction',
description: 'Phase 3 of the procurement-ingest refactor (docs/refactor-evaluation.md): extract lambdas/shared/ — ses_auth.py (byte-identical move, zero handler diff), web_ui_auth.py (byte-identical auth block from both web_ui handlers), email_parsing.py (parse_raw_email superset, cc unconditional), emf.py (parameterized emitter preserving every envelope byte-for-byte). Bundling gains cp shared/*.py in both email-processor commands + the same staging mechanism for both web_ui functions. Requires Phases 0-2 on the base branch. Auth code moves, so push is gated on /sh-security-review in the main loop. Committed locally, never pushed.',
phases: [
{ title: 'Setup', detail: 'verify Phases 0+1+2 on base, branch feature/phase-3-shared-extraction', model: 'haiku' },
{ title: 'Recon', detail: '4 mappers: the four duplicated modules, post-Phase-2 bundling sites, test-loader plumbing, deployed-zip baseline (read-only AWS)' },
{ title: 'Spec', detail: 'serial fable spec: pin shared-module contents, handler edit lists, web_ui staging, bundling strings, EMF design, test plumbing' },
{ title: 'Implement', detail: 'opus: shared/ + four handlers; sonnet: both CDK stacks; opus: all test plumbing + README — disjoint files', model: 'opus' },
{ title: 'Verify', detail: 'mechanical gates + bundle-parity verifier + 3 fable lenses (auth integrity, EMF/telemetry, loader integrity)' },
{ title: 'Fix', detail: 'opus fixer, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'single commit via -F (no push)', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-3-shared-extraction'
let _args = args
if (typeof _args === 'string') {
try { _args = JSON.parse(_args) } catch (e) { _args = null }
}
const BASE = (_args && _args.base) || 'main'
const CONSTRAINTS = `
PINNED BEHAVIORAL CONSTRAINTS (docs/refactor-evaluation.md Phase 3 — violating any is a build failure):
1. THE MOVE SET IS EXACTLY FOUR MODULES, in this order of dependency risk:
lambdas/shared/ses_auth.py, lambdas/shared/web_ui_auth.py,
lambdas/shared/email_parsing.py, lambdas/shared/emf.py. Nothing else
moves into shared/. lambdas/shared/ is the handbook location
(cdk-project-layout.md); modules land FLAT in every bundle so bare-name
imports keep working.
2. ses_auth: FIRST verify the two current copies are still byte-identical
(sha256 both against the ${BASE} versions — any drift since the audit is
a blocker, not something to silently reconcile). shared/ses_auth.py is
the EXACT bytes of that single copy; both originals are git rm'd. The
email-processor handlers keep 'from ses_auth import
authenticate_inbound_email' UNCHANGED — flat landing in /asset-output
means ZERO handler diff for this move, which is what keeps fail-closed
auth byte-identical through the change.
3. web_ui_auth: extract exactly the byte-identical block — _get_auth_token /
_header / is_authenticated + the four cache globals — from BOTH web_ui
handlers. The per-stack INFRA-74 comments STAY in each handler (their
wording has drifted deliberately; they are stack-specific — do NOT unify
or move them into the shared module). Fail-closed semantics (unset ARN,
Secrets Manager exception -> deny) must be unchanged; the token-cache
globals move with the functions that read them.
4. email_parsing.py: parse_raw_email as the SUPERSET version returning cc
unconditionally. WO's output is bit-identical to today; PO simply
ignores cc — do NOT "clean up" PO to consume it, and do NOT preserve two
variants.
5. emf.py: a generic emitter parameterized by namespace / dimension-sets /
properties. Every call site's emitted EMF envelope must be EXACTLY what
it emits today — the dimension-set list
[["ParseMethod"],["ParseMethod","TemplateId"]] is load-bearing for the
alarms and metric filters; making one-sided dimension fixes impossible
is the point of this move. Emission ORDERING is untouchable: PO emits
ai_fallback BEFORE the Bedrock call, WO after its gate with mutually-
exclusive ai_fallback/ai_fallback_rejected — these two deliberate
per-pipeline differences are pinned by tests; converting a call site
must not move it. The deliberate-double-count comments survive.
6. DERIVED-FIELDS EXCEPTION: if recon finds _emit_derived_agreement_metric
lives INSIDE derived_fields.py, do NOT convert it — derived_fields.py
and the shadow DerivedFieldAgreement telemetry are UNTOUCHABLE while the
bake runs (this outranks the emf consolidation). Leave it as a third
copy with a code comment pointing at shared/emf.py and report it in
notes/blockers. Only convert it if it lives outside derived_fields.py.
7. UNTOUCHABLE FILES (git diff ${BASE}...HEAD must be empty for each):
both template_parser.py (990 vs 508 lines, genuinely divergent — stays
per-pipeline), derived_fields.py, both validate_ai_fallback gate
modules, extract_with_claude's Bedrock invocation/prompt logic. Handler
diffs are LIMITED to: deleting moved code, import changes, and
emitter-call swaps per the binding spec. No opportunistic refactors.
8. Bundling (cdk): append 'cp shared/*.py /asset-output/' to BOTH
email-processor bundling commands (Phase 2 made the bundling cwd the
../lambdas asset root, so the path resolves as written).
BASE-AWARENESS — READ THE ACTUAL cdk FILES ON THE BASE, DO NOT ASSUME:
this phase is stacked on the Phase 7 branch, which ALREADY removed the
pip install step from both email-processor commands (they are now
CP-ONLY: no pip line, no manylinux pin, because nothing third-party is
installed). Phase 3 moves only pure first-party modules (no new deps),
so cp-only STAYS cp-only — you simply add another 'cp shared/*.py
/asset-output/' line. DO NOT re-introduce a pip install step and DO NOT
re-add the manylinux pin: there is nothing to pin when nothing installs,
and adding a pip step back would be the regression here. (The PR #34
lesson — never drop the manylinux pin while a pip install runs — still
holds ONLY if recon finds a pip step actually present on the base; on
the cp-only Phase 7 base there is none.) PRESERVE the Phase 2 exclude
lists exactly. Both web_ui functions gain the SAME staging mechanism
(widened-root bundled from_asset) so web_ui_auth.py ships beside their
handler — the spec agent pins the exact form. site_extractor's
from_asset is UNTOUCHED (Phase 6 territory).
9. tests/test_bundle_consistency.py is updated IN THE SAME CHANGE without
losing teeth: the PO_EXPECTED_TOP_LEVEL_MODULES exact-set pin gains the
shared modules that now ship; the AST sibling-import check must resolve
imports whose source file now lives under shared/; the new shared cp
line gets its own revert/mutation detection (a commented-out
'cp shared/*.py' must fail the test).
10. Test plumbing in the same PR: update _SIBLING_MODULES resolution and
_po_parser_support.py (~line 71 — verify the current line) so tests
load ses_auth/email_parsing/emf from shared/; retire or repoint the
fixture-hygiene test that polices the two ses_auth copies for
byte-identity (obsolete once there is one copy — do not leave it
failing); drop the ses_auth fixture params from test_ses_auth
(parameterizing over two identical copies is dead weight — halves the
run). The sys.modules save/restore dance SURVIVES for template_parser
(still a duplicated bare name) — do not delete it. Preserve the
load-bearing moto-before-handler import ordering.
11. STALE-SHADOW HAZARD: after the move, no stale ses_auth.py / .pyc /
__pycache__ copy may remain anywhere it could shadow the shared copy —
in the repo (git rm, don't empty), in staged assets, or on any test
sys.path. Verify explicitly.
12. cdk diff on BOTH stacks: the only resource deltas allowed are Code/
S3Key (asset) changes on the two email processors and the two web_ui
functions (+ CDK metadata). No logical-ID changes, no alarm, IAM,
table, env, runtime, or handler-property deltas of any kind.
13. AWS access is READ-ONLY (get-function, downloading deployed zips via
presigned URLs). NEVER cdk deploy, never invoke, never mutate.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 3 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push, do NOT run cdk deploy or
touch AWS resources beyond read-only calls).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 3".
Work ONLY in the files you are told you own; other agents are concurrently
editing other files in this same working tree.
${CONSTRAINTS}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const SPEC = {
type: 'object',
required: ['sharedModules', 'handlerEdits', 'webUiStaging', 'bundlingEdits', 'emfDesign', 'testPlumbing', 'parityRules', 'notes'],
properties: {
sharedModules: { type: 'string', description: 'per shared module: exact provenance (which copy is the source, sha256), full contents decision (verbatim move vs superset vs parameterized), and the public surface each importer uses' },
handlerEdits: { type: 'string', description: 'per handler (po/wo email_processor, po/wo web_ui): the exact deletions, import lines, and emitter-call swaps — file:line, nothing else may change' },
webUiStaging: { type: 'string', description: 'the complete new from_asset blocks for both web_ui functions: asset path, command or cp form, exclude list — exact code' },
bundlingEdits: { type: 'string', description: 'the two email-processor bundling command strings with cp shared/*.py appended, verbatim, plus the expected top-level module set of each resulting bundle' },
emfDesign: { type: 'string', description: 'shared/emf.py signature + per-call-site mapping proving each emitted envelope (namespace, dimension-set list, properties) is byte-equivalent to today; the derived-agreement emitter decision per constraint 6' },
testPlumbing: { type: 'string', description: 'every test/support file change: loader resolution, _SIBLING_MODULES, fixture-hygiene retirement, test_ses_auth de-parameterization, test_bundle_consistency edits with their mutation-detection shapes' },
parityRules: { type: 'string', description: 'how Verify judges staged bundles vs the deployed-zip baseline: expected first-party set per function (= deployed set + the new shared modules), ses_auth byte-identity requirement, expected removals (none beyond moved-file provenance), web_ui bundle expectations' },
notes: { type: 'string' },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const PARITY = {
type: 'object',
required: ['passed', 'poVerdict', 'woVerdict', 'webUiVerdict', 'details'],
properties: {
passed: { type: 'boolean' },
poVerdict: { type: 'string', description: 'PO email-processor staged vs deployed: module-set delta exactly as spec, ses_auth bytes identical — full evidence' },
woVerdict: { type: 'string', description: 'same for WO email-processor' },
webUiVerdict: { type: 'string', description: 'both web_ui staged bundles: own *.py + web_ui_auth.py present, nothing stray, hashes deterministic' },
details: { type: 'string' },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
// ------------------------------------------------------------------- setup
phase('Setup')
const setup = await agent(`
In ${REPO}:
1. SEQUENCING GATE — Phases 0, 1 AND 2 must all be on ${BASE} (Phase 3
edits the same stack files as Phase 2 and depends on its widened
../lambdas asset roots to make shared/ reachable). git fetch origin,
then pick the base ref: origin/${BASE} if that remote ref exists,
otherwise the local branch ${BASE} (a stacked local-only base is
expected and fine). Verify on the base ref:
(a) Phase 0: tests/test_bundle_consistency.py exists
(git show <baseref>:tests/test_bundle_consistency.py | head -3);
(b) Phase 1: validate_ai_fallback exists in the PO pipeline
(git grep validate_ai_fallback <baseref> -- lambdas/po);
(c) Phase 2: BOTH cdk/po_stack.py and cdk/wo_stack.py on the base ref
contain Code.from_asset("../lambdas") for the email processors
(git show <baseref>:cdk/po_stack.py | grep -n '\\.\\./lambdas', same
for wo_stack.py).
If any is missing, STOP with a blocker naming the unmet phase and do
nothing else.
2. Verify clean working tree (untracked .coverage / .claude/ / the local
44 MB lambdas/po/email_processor/package/ dir are fine; any OTHER dirt =
blocker, never stash or discard).
3. git checkout ${BASE}; then git pull --ff-only ONLY if the branch has an
upstream (a local-only base skips the pull — not a blocker); then
git checkout -b ${BRANCH}
4. gh pr list --state open --json number,title,headRefName (overlap check).
Return facts: HEAD sha, per-phase gate evidence, open PRs, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (!setup || (setup.blockers && setup.blockers.length)) {
return { aborted: 'setup blockers', blockers: setup ? setup.blockers : ['setup agent died'], facts: setup ? setup.facts : [] }
}
log(`Branch ${BRANCH} ready off ${BASE}. ${setup.summary}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of the FOUR duplicated surfaces being extracted:
1. ses_auth.py both copies — sha256 of each (MUST match; drift = blocker),
line count, the exact import line each email-processor handler uses,
any other importer (git grep 'import ses_auth\\|from ses_auth').
2. web_ui auth block — in BOTH web_ui handlers quote with file:line the
exact boundaries of the byte-identical block (_get_auth_token, _header,
is_authenticated, the four cache globals), diff the two blocks to prove
byte-identity, quote each INFRA-74 comment verbatim (they differ —
that is expected), and note everything else in each handler that CALLS
the block.
3. parse_raw_email both copies — where each lives (own sibling file vs
inside handler.py), the exact cc delta between them, every call site.
4. The THREE EMF emitters — quote each verbatim with file:line (PO
ParseMethod emitter, WO ParseMethod emitter,
_emit_derived_agreement_metric), namespace + dimension-set list +
properties of each, and CRITICALLY: which FILE _emit_derived_agreement_metric
lives in (constraint 6 hinges on whether it is inside derived_fields.py).
Also pin current line numbers of PO's pre-Bedrock ai_fallback emit and
WO's post-gate emit.
25-35 precise facts.`,
{ label: 'recon:duplicated-modules', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the post-Phase-2 CDK bundling state: for ALL FIVE
Code.from_asset sites in cdk/po_stack.py and cdk/wo_stack.py quote verbatim
with current file:line — asset path, full bundling command (or plain form),
exclude list. For the two web_ui functions additionally record: runtime,
architecture, handler property, memory/timeout, whether any requirements.txt
exists for them, and every function property that must NOT change when
bundling is added. Confirm lambdas/shared/ does not exist yet and list
anything at lambdas/ top level. 15-25 facts.`,
{ label: 'recon:cdk-bundling', model: 'haiku', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the test plumbing this phase must rewire:
- tests/conftest.py importlib loader: how it resolves module paths, the
sys.modules save/restore, _SIBLING_MODULES (exact current contents).
- _po_parser_support.py: the independent importlib reimplementation, the
load-bearing moto-before-handler import order (~lines 26-33), and what
is at line ~71 (the doc cites it — quote the current code).
- _wo_parser_support.py bare sys.path 'import handler' strategy.
- test_ses_auth.py at root: the fixture parameterization over both copies
(quote it), total line count, the fixture-hygiene test that polices
byte-identity between the two copies (name + assertion).
- test_parse_raw_email.py: how it loads both handlers' parse_raw_email.
- tests/test_bundle_consistency.py: every assertion that will fail red
against the Phase 3 shapes (shared cp line, PO_EXPECTED_TOP_LEVEL_MODULES,
AST sibling-import resolution for moved modules).
20-30 facts.`,
{ label: 'recon:test-plumbing', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only AWS recon (region us-east-1, READ-ONLY): for po-email-processor,
workorder-email-processor, AND both web_ui functions (find their exact
function names via the stacks/aws lambda list-functions) run aws lambda
get-function; for the two email processors download each Code.Location
presigned zip to a scratch dir and produce the COMPLETE file list
(unzip -l) separated into (a) first-party top-level .py, (b) dependency
dirs, (c) anything else; record every function's CodeSha256 and the
sha256 of ses_auth.py inside each deployed zip (the byte-identity baseline
the moved copy is judged against). For the web_ui functions the current
zip file list is the baseline their new bundled asset must cover.
Return the categorized lists as facts.`,
{ label: 'recon:deployed-baseline', model: 'sonnet', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n')
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
.filter(b => b && !/^\s*(none|n\/a)\b/i.test(b))
log(`Recon complete: ${reconOk.length}/4 mappers, ${reconBlockers.length} blockers`)
if (reconBlockers.length) {
return { aborted: 'recon blockers (likely copy drift — resolve before extracting)', blockers: reconBlockers, reconPack: pack }
}
// -------------------------------------------------------------------- spec
phase('Spec')
const spec = await agent(`${PREAMBLE}
You are the SPEC agent — the single authority that pins every contested
decision BEFORE parallel implementation (parallel leaves cannot see each
other's choices). Using the recon pack below plus your own reads of the
actual files, produce the binding implementation spec:
- sharedModules: per constraint 1's four modules — provenance, contents
decision, public surface. ses_auth is a verbatim byte-move; web_ui_auth
is the exact block; email_parsing is the WO superset (cc unconditional);
emf is the parameterized emitter.
- handlerEdits: for each of the four handlers, the exact minimal edit list
(constraint 7 — deletions, imports, emitter-call swaps ONLY). State
explicitly that the two email-processor ses_auth import lines are
UNCHANGED.
- webUiStaging: complete replacement from_asset blocks for both web_ui
functions — widened root, cp command staging the function's own *.py
plus web_ui_auth.py (pin whether to cp shared/*.py or just
shared/web_ui_auth.py — pick ONE rule, state why), Phase-2-style
excludes. Mind bundling cwd semantics: the container mounts the asset
root (../lambdas) as the working dir. If a no-Docker staging form is
viable and simpler, you may pin that instead — but ONE mechanism for
both, exact code.
- bundlingEdits: both email-processor command strings verbatim with
'cp shared/*.py /asset-output/' appended, plus the exact expected
top-level module set of each resulting bundle (this feeds
PO_EXPECTED_TOP_LEVEL_MODULES and the parity gate).
- emfDesign: the shared emitter signature and a per-call-site table
proving envelope byte-equivalence; resolve the derived-agreement emitter
per constraint 6 based on where recon found it.
- testPlumbing: every file, every edit — loader resolution for shared/,
_SIBLING_MODULES, _po_parser_support.py, _wo_parser_support.py if it
needs the shared path, fixture-hygiene retirement, test_ses_auth
de-parameterization, test_parse_raw_email (single implementation now —
what does it exercise?), test_bundle_consistency edits including the new
mutation shapes (commented-out shared cp must fail).
- parityRules: exactly how Verify judges staged bundles vs recon's
deployed baseline — per-function expected first-party set (deployed set
minus nothing, plus the shared modules per your cp rule), ses_auth
sha256 must equal the deployed zips' copy, web_ui bundles must cover
their deployed file list plus web_ui_auth.py, nothing else may appear
or vanish.
Recon pack:\n${pack}`,
{ label: 'spec:pin-extraction', phase: 'Spec', schema: SPEC })
if (!spec) return { aborted: 'spec agent died — rerun workflow', reconBlockers }
const specBlock = `BINDING SPEC (from the spec agent — implement EXACTLY this):\n${JSON.stringify(spec, null, 2)}`
log('Spec pinned: shared modules, handler edits, web_ui staging, bundling strings, EMF design, test plumbing')
// --------------------------------------------------------------- implement
phase('Implement')
const impl = await parallel([
() => agent(`${PREAMBLE}
YOU OWN: everything under lambdas/ EXCEPT the tests/ directories (another
agent owns all test and support files). Do not touch cdk/ or README.
Task: execute the four moves per spec.sharedModules + spec.handlerEdits +
spec.emfDesign, in the spec's order:
1. git mv (or create+git rm preserving exact bytes) ses_auth.py ->
lambdas/shared/ses_auth.py; delete both originals; prove sha256
equality in your summary. Handler import lines untouched.
2. Create lambdas/shared/web_ui_auth.py from the byte-identical block;
replace the block in BOTH web_ui handlers with the import; keep each
INFRA-74 comment in place.
3. Create lambdas/shared/email_parsing.py (superset); rewire both
email-processor call sites.
4. Create lambdas/shared/emf.py; swap the emitter call sites per
spec.emfDesign — honoring constraint 6 (derived_fields.py stays
untouched if the third emitter lives there) and constraint 5 (emission
ordering does not move).
Delete every now-empty moved-out file with git rm (constraint 11).
Run before returning: ruff check lambdas && ruff format lambdas --check,
plus an import smoke: python3 -c with sys.path prepended for
lambdas/shared + each function dir, importing every touched module (tests
are NOT yours to run — the plumbing agent lands them in parallel).
${specBlock}`,
{ label: 'impl:shared-lambdas', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: cdk/po_stack.py and cdk/wo_stack.py ONLY (and only the from_asset
regions — alarms, IAM, tables, env are all off-limits).
Task: apply spec.bundlingEdits (append the shared cp to both
email-processor commands — the base is the Phase 7 CP-ONLY bundling, so
just add another 'cp shared/*.py /asset-output/' line; DO NOT re-add a pip
install step or a manylinux pin, there is nothing third-party to install
(constraint 8); preserve the Phase 2 excludes verbatim) and
spec.webUiStaging (both web_ui functions). Touch nothing else;
site_extractor's from_asset stays byte-identical.
Run before returning: ruff check cdk && cd cdk &&
npx cdk synth po-ingest -q -o /tmp/phase3-synth &&
npx cdk synth workorder-ingest -q -o /tmp/phase3-synth (artifact-id
selectors, NOT stack_name). Docker bundling runs — confirm each staged
email-processor asset contains the shared modules and each staged web_ui
asset contains web_ui_auth.py; note asset hashes in your summary. If the
lambdas/ moves have not landed yet the synth will fail on missing files —
poll by re-running up to ~10 min before reporting a blocker.
${specBlock}`,
{ label: 'impl:cdk-stacks', model: 'sonnet', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: all test and support files (tests/ at repo root including
tests/test_bundle_consistency.py and tests/conftest.py,
lambdas/po/email_processor/tests/, lambdas/wo/email_processor/tests/)
and README.md ONLY.
Task A: apply spec.testPlumbing in full — loader/_SIBLING_MODULES
resolution for shared/, _po_parser_support.py edit (preserving the
moto-before-handler ordering comment), fixture-hygiene retirement,
test_ses_auth de-parameterization, test_parse_raw_email update,
test_bundle_consistency updates WITHOUT losing teeth (constraint 9 — the
new shared cp pin must reject a commented-out or narrowed variant; keep
the existing mutation tests green).
Task B: README — document lambdas/shared/ in the repo-layout section
(which modules live there, the flat-landing import rule, the bundling cp
that ships them), update the bundling/deploy-guards paragraphs for the
shared cp + web_ui staging, and note that ses_auth is now single-sourced
(one hardening fix lands once). Match existing README style.
Run before returning: pytest -q --no-cov at repo root (must be green
against the OTHER agents' edits — they land in parallel; poll by
re-running up to ~10 min before reporting a blocker) and ruff check on
every file you touched.
${specBlock}`,
{ label: 'impl:test-plumbing-readme', model: 'opus', phase: 'Implement', schema: IMPL }),
])
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/3 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'lambdas/shared/',
'lambdas/po/email_processor/',
'lambdas/wo/email_processor/',
'lambdas/po/web_ui/',
'lambdas/wo/web_ui/',
'cdk/po_stack.py',
'cdk/wo_stack.py',
'tests/',
'README.md',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification — trust nothing self-reported. Run ALL gates,
quoting failures verbatim:
1. pytest -q --no-cov (repo root, all three roots green)
2. ruff check . && ruff format --check .
3. cd cdk && npx cdk synth po-ingest -q && npx cdk synth workorder-ingest -q
4. cd cdk && npx cdk diff po-ingest ; npx cdk diff workorder-ingest —
constraint 12: only the four functions' Code/S3Key (+ metadata) deltas.
Any alarm/IAM/env/runtime/handler-prop/logical-ID delta = FAIL. Paste
the diff summaries.
5. UNTOUCHABLES (each must output NOTHING):
git diff ${BASE}...HEAD -- lambdas/po/email_processor/derived_fields.py
git diff ${BASE}...HEAD -- lambdas/po/email_processor/template_parser.py
git diff ${BASE}...HEAD -- lambdas/wo/email_processor/template_parser.py
plus both validate_ai_fallback gate modules (resolve their filenames
first) and lambdas/po/site_extractor/.
6. BYTE-IDENTITY: sha256 of lambdas/shared/ses_auth.py equals sha256 of
git show ${BASE}:lambdas/po/email_processor/ses_auth.py (and the wo
copy). Both originals gone from the tree (git ls-files check) and no
stray ses_auth.py/__pycache__ anywhere under lambdas/ outside shared/
(constraint 11).
7. ORDERING PINS: grep line numbers proving PO's ai_fallback emit still
precedes the Bedrock invoke and WO's emits are untouched relative to
${BASE} (the wo handler diff must contain ONLY the spec's edit classes).
8. INFRA-74: both web_ui handlers still contain their own INFRA-74
comment verbatim per ${BASE}.
9. git status --porcelain scope check: every modified/added/deleted path
under ${EXPECTED_SCOPE.join(', ')} (untracked .coverage/.claude/
package/ tolerated).
passed=true only if all green. YOU MAY NOT edit files.`
const parityPrompt = `${PREAMBLE}
You are the BUNDLE-PARITY verifier — the load-bearing gate of this phase.
Everything is local synth + read-only AWS.
1. cd cdk && npx cdk synth po-ingest -q -o /tmp/phase3-parity &&
npx cdk synth workorder-ingest -q -o /tmp/phase3-parity. Locate all
four staged assets (two email processors, two web_ui).
2. Re-download both email-processor deployed zips fresh (aws lambda
get-function Code.Location, us-east-1) — do not trust a recon cache —
and fetch both web_ui functions' deployed file lists.
3. Judge per spec.parityRules: per email processor, staged first-party
top-level .py set == deployed set PLUS exactly the shared modules the
spec's cp rule ships (list both sets; nothing else may appear or
vanish); sha256 of staged ses_auth.py == sha256 of the ses_auth.py
inside each deployed zip (fail-closed auth byte-identical through the
move); dependency packages compared by name, version drift noted not
failed. Per web_ui function: staged bundle covers the deployed file
list plus web_ui_auth.py, nothing stray (no tests/, no other
pipeline's sources, no .eml, no package/).
4. DETERMINISM: synth po-ingest twice into fresh -o dirs — identical
asset hashes, including the NEW web_ui assets. Then drop a throwaway
__pycache__/junk.pyc under lambdas/shared/ (delete it afterwards),
re-synth, and confirm the excludes keep every hash unchanged.
5. IMPORT-RESOLUTION sanity: inside each staged email-processor asset
dir run python3 -c "import handler" with that dir alone on sys.path
(env-var stubs as needed) — proves the flat landing satisfies every
import including 'from ses_auth import ...' with the moved copy.
passed=true only if every check holds.`
const lenses = [
{ key: 'auth-integrity', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — auth-integrity lens. Auth code moved; try to prove
the move WEAKENED it. (1) ses_auth: byte-compare the shared copy against
${BASE}'s copies yourself; then attack import resolution — in the BUNDLE
and in TESTS, which ses_auth wins if anything shadows (stale .pyc, a
same-named module on sys.path, the tests loader resolving the old path
silently to a stub)? Could a test now pass against a MOCK of ses_auth
where it previously exercised the real module? (2) web_ui_auth: diff the
extracted block against both originals — any dropped line, changed
global, or reordered check? Is the fail-closed path (unset ARN, Secrets
exception, wrong token -> 401 before any table access) provably
unchanged? Do the four cache globals still behave per-function (module
now shared — could cross-importer state ever leak)? (3) The gate call
sites: could any handler path now reach S3-fetch/Bedrock/save before
authenticate_inbound_email or is_authenticated, where it could not
before? confirmed=true only with a concrete exploit sketch or
file:line proof.` },
{ key: 'telemetry-emf', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — EMF/telemetry lens. The alarms and metric filters
consume exact EMF shapes; a silent envelope change breaks paging without
failing any test. Read shared/emf.py + every converted call site
(git diff ${BASE}) and the synthesized templates' metric filters/alarms.
Verify per call site: namespace exact, dimension-set list EXACT
([["ParseMethod"],["ParseMethod","TemplateId"]] where applicable, order
included), property keys/types identical, timestamp/CloudWatchMetrics
envelope structure identical — construct a sample emission per call site
and diff it against ${BASE}'s hand-built _aws dict output. Then the
ordering pins: PO ai_fallback still pre-Bedrock, WO still post-gate
mutually-exclusive, the deliberate double-count comment intact. Finally
constraint 6: where does _emit_derived_agreement_metric live and was the
decision honored (derived_fields.py diff empty)? confirmed=true only
with evidence.` },
{ key: 'loader-integrity', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — test/loader-integrity lens. The three loading idioms
were rewired; attack them. (1) Does test_ses_auth now exercise the REAL
lambdas/shared/ses_auth.py (trace the loader path), and did
de-parameterization silently drop any assertion that only ran under one
param? (2) Fixture-hygiene: the byte-identity police is retired — does
anything still guard against a future stray ses_auth.py copy reappearing
in a pipeline dir (should the bundle-consistency or a hygiene test)? If
nothing does, that is a finding. (3) sys.modules save/restore: prove it
still isolates template_parser between pipelines (run two cross-pipeline
tests back to back); prove moto-before-handler ordering survived in
_po_parser_support.py. (4) test_bundle_consistency: hand-mutate command
strings on scratch copies — a commented-out 'cp shared/*.py', a shared
glob narrowed to one file, and a PO_EXPECTED_TOP_LEVEL_MODULES missing a
shared module must each FAIL. (5) Run pytest twice in one session and in
file-shuffled order (-p no:randomly not installed? then two explicit
orderings) to smoke out import-order coupling introduced by the shared
path. confirmed=true only with file:line or reproduced-failure
evidence.` },
]
let round = 0
let checks = null
let parity = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
() => agent(parityPrompt, { label: `verify:parity-r${round}`, model: 'opus', phase: 'Verify', schema: PARITY }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
parity = results[1]
confirmed = results.slice(2).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && f.severity !== 'low')
const green = checks && checks.passed && parity && parity.passed
log(`Verify round ${round}: mechanical ${checks && checks.passed ? 'GREEN' : 'RED'}, parity ${parity && parity.passed ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (green && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit files under: ${EXPECTED_SCOPE.join(', ')}.
Fix EVERY item below minimally; the binding spec and 13 pinned constraints
still hold (a finding that conflicts with a constraint is reported, not
"fixed" — the constraint wins, esp. constraint 6's derived_fields
untouchability and constraint 5's emission ordering). Re-run the specific
failing gate/test per fix.
MECHANICAL:\n${checks ? checks.details : '(agent died — rerun all gates)'}
PARITY:\n${parity ? parity.details : '(agent died — rerun all)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}
${specBlock}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && parity && parity.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
parity,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
spec,
}
}
// ----------------------------------------------------------------- package
phase('Package')
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent:
1. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md
and follow it exactly.
2. git add only paths under: ${EXPECTED_SCOPE.join(', ')} and
.claude/workflows/phase-3-shared-extraction.js. NOT .coverage, NOT
package/. Verify the staged set with git status — the deletions of the
moved originals MUST be staged too.
3. ONE commit; write the message to /tmp/phase3-commit-msg.txt and use
git commit -F /tmp/phase3-commit-msg.txt (backticks in -m get eaten by
zsh). Suggested subject:
"feat: extract lambdas/shared/ — single-source ses_auth, web_ui auth, email parsing, EMF emitter (refactor phase 3)"
Body: the four moves with the byte-identity evidence one-liner for
ses_auth, the flat-landing import rule, the shared cp bundling change +
web_ui staging, the derived-agreement emitter decision (constraint 6),
and the test-plumbing summary. NO AI attribution / Co-Authored-By
lines.
4. Do NOT push. Return commit sha + shortstat in summary.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
base: BASE,
commit: commit ? commit.summary : 'commit agent died — commit manually',
spec: { sharedModules: spec.sharedModules, webUiStaging: spec.webUiStaging, emfDesign: spec.emfDesign, derivedAgreementNote: spec.notes },
parityEvidence: parity ? { po: parity.poVerdict, wo: parity.woVerdict, webUi: parity.webUiVerdict } : null,
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
outstandingGates: [
'/sh-security-review (MANDATORY before push — auth code moved: ses_auth + the web_ui auth gate are exactly the authentication surface; run on the committed diff, and re-run after any post-review fix to this code)',
'cross-family cross_review.py NOT mandatory (no IAM change; handler event/return contracts unchanged — internal module moves only). Opt-in if judgment says so',
'push + PR + gh pr checks green',
'deploy-then-merge: deploy from branch, smoke green, one real PO + WO email each, watch BOTH sender-auth-rejected alarms through live mail (the auth move must not change accept/reject behavior), verify web_ui login still works, THEN merge — and note the post-merge CI redeploy being a no-op (unchanged asset hashes) is itself a verification signal',
],
}

View file

@ -1,684 +0,0 @@
export const meta = {
name: 'phase-4-cdk-common',
description: 'Phase 4 of the procurement-ingest refactor (docs/refactor-evaluation.md): collapse the 379 identical CDK lines into cdk/common.py as PLAIN FUNCTIONS taking (scope, id, ...) — called with the SAME Stack scope and the SAME construct ids the stacks use today, so every logical ID is byte-stable (Construct-subclass wrapping is forbidden: it would reparent the tree and attempt REPLACEMENT of the RETAIN-protected purchase-orders/WorkOrders tables and named buckets = data loss). Extracts add_ddb_alarms, add_sender_auth_rejected_alarm, add_standard_lambda_alarms, make_bedrock_invoke_statement (account/region-derived ARN, not hardcoded 328440206208), make_email_bucket, make_processor_dlq, make_fallback_rate_alarm — preserving every per-function alarm variance byte-for-byte. Same PR: account on both cdk.Environment, constructs== exact pin, stale-comment fix, CfnOutputs for five function ARNs + consumed table names. ZERO cdk diff on both stacks is the acceptance test. The make_bedrock_invoke_statement IAM PolicyStatement move triggers mandatory GPT-4.1 cross-family review even though semantics are identical. Committed locally, never pushed.',
phases: [
{ title: 'Setup', detail: 'verify Phases 0+1+2+3 on base, branch feature/phase-4-cdk-common', model: 'haiku' },
{ title: 'Recon', detail: '4 mappers: the 379 duplicated CDK lines + per-function alarm variance, the two inline Bedrock PolicyStatements, the fallback-rate + rejected alarm math (post-Phase-1), app.py/pins/outputs surface' },
{ title: 'Spec', detail: 'serial fable spec: pin common.py contents + every function signature, per-stack call-site rewrites, alarm-variance table, Bedrock IAM equivalence, fallback-rate call params, app/pins/CfnOutputs, zero-diff judging rules' },
{ title: 'Implement', detail: 'opus: cdk/common.py; opus: both stacks (rewire + CfnOutputs); sonnet: app.py + requirements pin + README — disjoint files', model: 'opus' },
{ title: 'Verify', detail: 'mechanical gates + zero-cdk-diff verifier (the load-bearing gate) + 3 fable lenses (logical-ID safety, alarm variance, IAM equivalence)' },
{ title: 'Fix', detail: 'opus fixer, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'single commit via -F (no push); runs cross_review.py inline on the IAM diff', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-4-cdk-common'
let _args = args
if (typeof _args === 'string') {
try { _args = JSON.parse(_args) } catch (e) { _args = null }
}
const BASE = (_args && _args.base) || 'main'
const CONSTRAINTS = `
PINNED BEHAVIORAL CONSTRAINTS (docs/refactor-evaluation.md Phase 4 — violating any is a build failure):
1. EXTRACT AS PLAIN FUNCTIONS taking (scope, id, ...), called with the SAME
scope (the Stack instance) and the SAME construct ids the stacks use
today -> 100% logical-ID-safe. NEVER wrap in Construct subclasses: a
subclass inserts a tree node, changes EVERY child logical ID, and would
attempt REPLACEMENT of the RETAIN-protected purchase-orders / WorkOrders
tables and the named buckets = DATA LOSS. This is THE load-bearing rule
of the phase — every other check exists to defend it.
2. New module cdk/common.py. Extract exactly:
- _DDB_ALARM_OPERATIONS + add_ddb_alarms
- add_sender_auth_rejected_alarm
- add_standard_lambda_alarms(scope, id_prefix, fn, name_prefix, topic, *,
duration_statistic, errors=True, dlq=None, descriptions=...)
- make_bedrock_invoke_statement (DERIVE the inference-profile ARN + the
per-region foundation-model ARNs from Stack.of(scope).account /
Stack.of(scope).region — NOT hardcoded 328440206208)
- make_email_bucket
- make_processor_dlq
- make_fallback_rate_alarm(namespace, rejected_included, period, threshold,
floor, evaluation_periods, datapoints_to_alarm) reproducing the
expression strings / FILL / labels BYTE-FOR-BYTE.
3. PER-FUNCTION ALARM VARIANCE — preserve EXACTLY, do NOT homogenize:
PO email-processor duration p99 vs wo-email-processor p95; po-web-ui
throttles+duration only; site_extractor no-DLQ; wo web_ui has ZERO alarms
(do NOT let the shared helper silently add any); every bespoke alarm
DESCRIPTION string is passed through verbatim.
4. FALLBACK-RATE RECONCILIATION (post-Phase-1): PO's template-fallback-rate
EXCLUDES the rejected series (byte-identical to today, a pre-call
double-count would result otherwise) -> call make_fallback_rate_alarm with
rejected_included=False; WO's INCLUDES rejected -> rejected_included=True.
The two REJECTED alarms are a DIFFERENT shape and are NOT
make_fallback_rate_alarm: PO's ai-fallback-rejected is a 6h count-floor
IF(FILL(rej,0)>=1,...) alarm (added in Phase 1); WO's is the 5-min /
2-of-6 sparse idiom. Keep those as DISTINCT call sites (or a separate
dedicated helper) — do NOT force them through make_fallback_rate_alarm.
NO element-wise MAX anywhere (post-#102 rule).
5. Do NOT import stack-specific services (kms / ssm / event_sources) into
common.py — only the constructs the shared helpers actually need.
6. SAME PR, net-new & logical-ID-safe additions:
- add account='328440206208' to BOTH cdk.Environment calls
- pin constructs== to the exact installed version (not a floor >=)
- fix the stale "2.259.0" version comments
- add CfnOutputs for the FIVE function ARNs + the consumed table names.
These are additive; CfnOutputs and account are ID-safe. Verify none of
them perturbs an existing logical ID.
7. ZERO lambdas/ diff: git diff ${BASE}...HEAD -- lambdas/ must be EMPTY.
This phase is CDK-ONLY. tests/ may gain a cdk-diff / synth-only test for
the acceptance gate, but NO other tests/ change and NO lambdas/ change.
8. ZERO cdk diff on BOTH stacks is the acceptance test: npx cdk diff
po-ingest and npx cdk diff workorder-ingest must show ZERO resource
changes (no logical-ID, alarm, IAM, table, bucket, env, or metadata
delta beyond CDK-tooling noise). The CfnOutputs are the ONLY net-new
resources allowed to appear, and only as additions.
9. The wo artifact id is 'workorder-ingest' (the construct id / 2nd
positional arg), NOT 'WorkorderIngestStack' (that is stack_name). ALWAYS
drive synth/diff by the artifact id: npx cdk synth workorder-ingest,
npx cdk diff workorder-ingest. Using the stack_name selector fails.
10. NO cdk deploy, NO invoke, NO AWS mutation. Read-only AWS only if needed
(e.g. confirming deployed alarm names) — the diff gate is a pure local
synth-vs-synth comparison.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 4 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push, do NOT run cdk deploy or
touch AWS resources beyond read-only calls).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 4".
Work ONLY in the files you are told you own; other agents are concurrently
editing other files in this same working tree.
${CONSTRAINTS}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const SPEC = {
type: 'object',
required: ['commonModule', 'poStackEdits', 'woStackEdits', 'alarmVariance', 'bedrockStatement', 'fallbackRateCalls', 'appAndPins', 'diffRules', 'notes'],
properties: {
commonModule: { type: 'string', description: 'the full cdk/common.py: every function (add_ddb_alarms + _DDB_ALARM_OPERATIONS, add_sender_auth_rejected_alarm, add_standard_lambda_alarms, make_bedrock_invoke_statement, make_email_bucket, make_processor_dlq, make_fallback_rate_alarm) with its EXACT signature, and the exact imports it needs (no stack-specific kms/ssm/event_sources per constraint 5)' },
poStackEdits: { type: 'string', description: 'cdk/po_stack.py: every inline block replaced by a common.* call, file:line, with the exact scope + construct-id + kwargs each call passes so the emitted resource is byte-identical; plus the PO CfnOutput additions (function ARNs + consumed table names)' },
woStackEdits: { type: 'string', description: 'cdk/wo_stack.py: same — call-site rewrites file:line preserving construct ids, plus WO CfnOutput additions; explicitly note wo web_ui gets NO alarms (constraint 3)' },
alarmVariance: { type: 'string', description: 'the per-function alarm-variance table proving each helper call reproduces exactly what the inline code emits today: PO p99 vs wo-email-processor p95, po-web-ui throttles+duration only, site_extractor no-DLQ, wo web_ui ZERO alarms, every bespoke description string mapped verbatim' },
bedrockStatement: { type: 'string', description: 'make_bedrock_invoke_statement: the exact actions + resources, showing how the inference-profile ARN and per-region FM ARNs are derived from Stack.of(scope).account/.region, and PROVING the derived strings resolve in-account to the SAME ARNs the two inline PolicyStatements hardcode today' },
fallbackRateCalls: { type: 'string', description: 'the make_fallback_rate_alarm call params for PO (rejected_included=False) and WO (rejected_included=True) reproducing the expression/FILL/label strings byte-for-byte; PLUS how the two DISTINCT rejected alarms stay distinct call sites (PO 6h count-floor IF(FILL(rej,0)>=1,...); WO 5-min/2-of-6 sparse) — NOT folded into make_fallback_rate_alarm, NO element-wise MAX' },
appAndPins: { type: 'string', description: 'app.py account= additions to both cdk.Environment calls; the exact constructs== pin (installed version); the stale "2.259.0" comment fix locations + new text; confirmation none perturbs a logical ID' },
diffRules: { type: 'string', description: 'exactly how Verify proves zero cdk diff: synth BASE and HEAD into separate temp dirs, diff the two stacks templates, ANY resource/logical-ID/property delta = FAIL, the ONLY allowed additions are the net-new CfnOutputs' },
notes: { type: 'string' },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const DIFF = {
type: 'object',
required: ['passed', 'poDiffVerdict', 'woDiffVerdict', 'details'],
properties: {
passed: { type: 'boolean' },
poDiffVerdict: { type: 'string', description: 'po-ingest BASE-synth vs HEAD-synth: ZERO resource/logical-ID/property changes (CfnOutputs the only allowed net-new additions) — full template-diff evidence' },
woDiffVerdict: { type: 'string', description: 'workorder-ingest (artifact id, NOT WorkorderIngestStack): same zero-change evidence' },
details: { type: 'string' },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
// ------------------------------------------------------------------- setup
phase('Setup')
const setup = await agent(`
In ${REPO}:
1. SEQUENCING GATE — Phases 0, 1, 2 AND 3 must all be on ${BASE} (Phase 4
dedups BOTH cdk stacks, which Phases 1 (po_stack alarms), 2 (bundling
roots) and 3 (shared cp) all edited — building against a pre-Phase-3
stack file guarantees a conflict and a wrong diff baseline). git fetch
origin, then pick the base ref: origin/${BASE} if that remote ref
exists, otherwise the local branch ${BASE} (a stacked local-only base is
expected and fine). Verify on the base ref:
(a) Phase 3: lambdas/shared/ exists
(git ls-tree <baseref> -- lambdas/shared | head);
(b) Phase 2: BOTH cdk/po_stack.py and cdk/wo_stack.py contain
Code.from_asset("../lambdas") for the email processors
(git show <baseref>:cdk/po_stack.py | grep -n '\\.\\./lambdas', same
for wo_stack.py);
(c) Phase 1: the po-email-processor-ai-fallback-rejected alarm /
EmailProcessorAiFallbackRejectedAlarm construct exists in po_stack.py
(git show <baseref>:cdk/po_stack.py | grep -n 'ai-fallback-rejected\\|AiFallbackRejected').
If Phase 3 is not on ${BASE}, STOP with a blocker (Phase 4 needs the
post-Phase-3 stack files as its zero-diff baseline). If any other phase
is missing, STOP with a blocker naming the unmet phase and do nothing
else.
2. Verify clean working tree (untracked .coverage / .claude/ / the local
lambdas/po/email_processor/package/ dir are fine; any OTHER dirt =
blocker, never stash or discard).
3. git checkout ${BASE}; then git pull --ff-only ONLY if the branch has an
upstream (a local-only base skips the pull — not a blocker); then
git checkout -b ${BRANCH}
4. gh pr list --state open --json number,title,headRefName (overlap check).
Return facts: HEAD sha, per-phase gate evidence, open PRs, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (!setup || (setup.blockers && setup.blockers.length)) {
return { aborted: 'setup blockers', blockers: setup ? setup.blockers : ['setup agent died'], facts: setup ? setup.facts : [] }
}
log(`Branch ${BRANCH} ready off ${BASE}. ${setup.summary}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of the ~379 duplicated CDK lines and the PER-FUNCTION ALARM
VARIANCE that MUST survive the dedup (constraint 3). In cdk/po_stack.py and
cdk/wo_stack.py, quote verbatim with file:line:
1. _DDB_ALARM_OPERATIONS + add_ddb_alarms (both copies — are they
byte-identical? diff them).
2. add_sender_auth_rejected_alarm (both copies).
3. Every add_standard_lambda_alarms-shaped block: for EACH function
(po email-processor, wo email-processor, po web_ui, wo web_ui,
po site_extractor) list the exact alarm set, the duration statistic
(prove PO email p99 vs wo email p95), whether throttles/errors/dlq
alarms are present, and QUOTE every bespoke alarm description string.
CRITICALLY: confirm wo web_ui has ZERO alarms today.
4. make_email_bucket / make_processor_dlq shaped blocks (both copies), and
which functions get a DLQ (site_extractor has none).
5. The exact construct ids (2nd positional arg to every alarm / bucket /
dlq / statement construct) — these are the logical-ID roots that must be
passed UNCHANGED into the shared helpers.
30-40 precise facts. Any block that is NOT actually identical between
stacks (genuine drift) is a blocker to report, not to silently reconcile.`,
{ label: 'recon:duplicated-cdk', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the two inline Bedrock IAM PolicyStatements (the
make_bedrock_invoke_statement source — constraint 2, the cross-family-review
surface). In both stacks quote verbatim with file:line: the full
iam.PolicyStatement (effect, actions, resources, conditions). For EACH
resource ARN record whether the account (328440206208) and region are
hardcoded or referenced, and enumerate every inference-profile ARN and every
per-region foundation-model ARN. Determine the EXACT list of regions / model
ids baked into the resources so the derived form (Stack.of(scope).account /
.region) can be proven to resolve to the identical strings in-account. Note
which principal/role each statement is attached to and how (add_to_role_policy
vs inline policy). 15-25 facts.`,
{ label: 'recon:bedrock-iam', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the fallback-rate + rejected alarm math AS IT STANDS
POST-PHASE-1 (constraint 4). In both stacks quote verbatim with file:line:
1. PO's template-fallback-rate alarm: the full metric-math expression
string(s), every FILL(), every label, the period/threshold/
evaluation_periods/datapoints_to_alarm — and CONFIRM it EXCLUDES the
rejected series today (rejected_included=False).
2. WO's template-fallback-rate alarm: same, and CONFIRM it INCLUDES the
rejected series (rejected_included=True).
3. PO's ai-fallback-rejected alarm (added in Phase 1): confirm it is the
6h count-floor IF(FILL(rej,0)>=1,...) shape — quote the expression.
4. WO's rejected alarm: confirm it is the 5-min / 2-of-6 sparse idiom —
quote it. These two are DIFFERENT shapes and must stay distinct call
sites, NOT folded into make_fallback_rate_alarm.
5. Confirm NO element-wise MAX exists anywhere in either expression
(post-#102 rule).
Return the exact parameter values each make_fallback_rate_alarm call must
carry so Verify can prove byte-identity. 15-25 facts.`,
{ label: 'recon:fallback-alarms', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the app.py / pins / outputs surface (constraint 6):
1. cdk/app.py: quote both cdk.Environment(...) calls verbatim with line
numbers (region-only today; account must be added).
2. The constructs dependency pin: quote cdk/requirements.txt line(s) and
find the stale "2.259.0" version comment(s) wherever they live (app.py,
stacks, requirements — grep the whole cdk/ tree) with file:line and the
actual installed constructs / aws-cdk-lib versions.
3. The five Lambda function construct variables in the stacks whose ARNs
need CfnOutputs (po email-processor, po web_ui, po site_extractor,
wo email-processor, wo web_ui) and the table constructs whose names are
consumed (purchase-orders, WorkOrders, and any comments table) — quote
the construct handles + ids so the CfnOutputs reference them correctly.
4. Any existing CfnOutput in either stack (WO reportedly has zero).
5. Confirm the wo artifact id is 'workorder-ingest' (the 2nd positional
arg to the Stack constructor in app.py), distinct from
stack_name='WorkorderIngestStack' (constraint 9).
15-25 facts.`,
{ label: 'recon:app-pins-outputs', model: 'haiku', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n')
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
.filter(b => b && !/^\s*(none|n\/a)\b/i.test(b))
log(`Recon complete: ${reconOk.length}/4 mappers, ${reconBlockers.length} advisory notes`)
// Recon "blockers" for this phase are advisory design-notes / intended
// constraint-2 & 6 work items (derive the Bedrock ARN from Stack.of(scope),
// add account= to the environments, exact-pin constructs, fix the stale
// aws-cdk-lib version comment, decide the one common sender-auth docstring),
// NOT stop conditions — main-loop verified. The real gate is the ZERO cdk diff
// acceptance test in Verify. Fold the notes into the spec context instead of
// aborting so the spec agent must address each.
const reconAdvisories = reconBlockers.length
? `\n\nRECON ADVISORIES (recon flagged these; they are the intended constraint-2/6 work + a docstring choice, NOT drift that blocks dedup — resolve each per the constraints; the zero-cdk-diff test is the real acceptance gate):\n- ${reconBlockers.join('\n- ')}`
: ''
// -------------------------------------------------------------------- spec
phase('Spec')
const spec = await agent(`${PREAMBLE}
You are the SPEC agent — the single authority that pins every contested
decision BEFORE parallel implementation (parallel leaves cannot see each
other's choices). Using the recon pack below plus your own reads of the
actual files, produce the binding implementation spec:
- commonModule: the full cdk/common.py — every function per constraint 2
with its EXACT signature (add_standard_lambda_alarms keyword-only params
exactly as the doc pins them), and ONLY the imports the helpers need
(constraint 5 — no kms/ssm/event_sources). Every function is a PLAIN
function taking (scope, id, ...) — NO Construct subclass anywhere
(constraint 1).
- poStackEdits / woStackEdits: for each stack, the exact call-site rewrites
(file:line) mapping each inline block to a common.* call, passing the
SAME scope (the Stack) and the SAME construct id it uses today so every
logical ID is byte-stable; plus the CfnOutput additions (five function
ARNs split across the two stacks + consumed table names). State
explicitly that wo web_ui receives NO alarm call (constraint 3).
- alarmVariance: the per-function variance table (constraint 3) proving each
helper call reproduces today's emitted alarms EXACTLY — PO p99 vs wo-email
p95, po-web-ui throttles+duration only, site_extractor no-DLQ, wo web_ui
zero alarms, every bespoke description string mapped verbatim.
- bedrockStatement: make_bedrock_invoke_statement's actions + resources and
the derivation of the inference-profile / per-region FM ARNs from
Stack.of(scope).account/.region, with a proof table showing each derived
string equals the inline hardcoded ARN in-account (this is the
cross-family-review surface — be exhaustive).
- fallbackRateCalls: the make_fallback_rate_alarm call params for PO
(rejected_included=False) and WO (rejected_included=True) reproducing the
expression/FILL/label strings byte-for-byte, PLUS the plan for keeping the
two DISTINCT rejected alarms as separate call sites (PO 6h count-floor,
WO 5-min/2-of-6) — NOT folded, NO element-wise MAX (constraint 4).
- appAndPins: app.py account= on both Environment calls; the exact
constructs== pin; the stale "2.259.0" comment fix (file:line + new text);
confirmation each is logical-ID-neutral.
- diffRules: exactly how Verify proves ZERO cdk diff — synth ${BASE} and
HEAD each into a separate temp dir, diff both stacks' templates, ANY
resource/logical-ID/property delta = FAIL, the ONLY allowed net-new is
the CfnOutputs; drive synth/diff by artifact id (po-ingest /
workorder-ingest, constraint 9).
Recon pack:\n${pack}${reconAdvisories}`,
{ label: 'spec:pin-common', phase: 'Spec', schema: SPEC })
if (!spec) return { aborted: 'spec agent died — rerun workflow', reconBlockers }
const specBlock = `BINDING SPEC (from the spec agent — implement EXACTLY this):\n${JSON.stringify(spec, null, 2)}`
log('Spec pinned: common.py contents, per-stack rewrites, alarm variance, Bedrock IAM, fallback-rate calls, app/pins/outputs')
// --------------------------------------------------------------- implement
phase('Implement')
const impl = await parallel([
() => agent(`${PREAMBLE}
YOU OWN: cdk/common.py ONLY (create it). Do not touch po_stack.py,
wo_stack.py, app.py, requirements.txt, README, tests, or anything under
lambdas/.
Task: author cdk/common.py per spec.commonModule EXACTLY — every function
(add_ddb_alarms + _DDB_ALARM_OPERATIONS, add_sender_auth_rejected_alarm,
add_standard_lambda_alarms with the pinned keyword-only signature,
make_bedrock_invoke_statement deriving ARNs from Stack.of(scope).account/
.region, make_email_bucket, make_processor_dlq, make_fallback_rate_alarm)
as a PLAIN function taking (scope, id, ...) — NEVER a Construct subclass
(constraint 1). Import ONLY what the helpers need — no kms/ssm/
event_sources (constraint 5). Match the fallback-rate expression/FILL/label
strings byte-for-byte (constraint 4). Do NOT put the two rejected alarms in
make_fallback_rate_alarm.
Run before returning: ruff check cdk/common.py && ruff format cdk/common.py
--check, plus a py_compile import smoke (python3 -c "import common" from
cdk/ with the CDK venv). Full synth is the stacks agent's job — but if you
can import common cleanly, report it.
${specBlock}`,
{ label: 'impl:common-module', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: cdk/po_stack.py and cdk/wo_stack.py ONLY. Do not touch
cdk/common.py (another agent authors it), app.py, requirements.txt, README,
or lambdas/.
Task: apply spec.poStackEdits + spec.woStackEdits — replace each inline
block with the matching common.* call, passing the SAME scope (the Stack
instance, NOT a new Construct) and the SAME construct id used today so every
logical ID is byte-stable (constraint 1). Preserve every per-function alarm
variance (constraint 3): PO email p99 / wo email p95, po-web-ui throttles+
duration only, site_extractor no-DLQ, wo web_ui gets NO alarm call, bespoke
descriptions passed verbatim. Wire the fallback-rate calls per
spec.fallbackRateCalls (PO rejected_included=False, WO True) and keep the
two rejected alarms as distinct call sites (constraint 4). Add the CfnOutputs
per spec (function ARNs + consumed table names) — additive, ID-safe.
Import from common (bare 'import common' / 'from common import ...' — cdk/
is on sys.path via app.py's imports). Touch nothing else (tables, KMS, SSM,
event sources, the RETAIN policies stay byte-identical).
Run before returning: ruff check cdk && cd cdk &&
npx cdk synth po-ingest -q -o /tmp/phase4-synth &&
npx cdk synth workorder-ingest -q -o /tmp/phase4-synth (artifact-id
selectors, NOT stack_name — constraint 9). If cdk/common.py has not landed
yet the synth fails on the missing import — poll by re-running up to ~10 min
before reporting a blocker. Confirm both stacks synth and note that the
ONLY template delta vs ${BASE} is the net-new CfnOutputs (spot-check a
couple of alarm logical IDs are unchanged).
${specBlock}`,
{ label: 'impl:cdk-stacks', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: cdk/app.py, cdk/requirements.txt and README.md ONLY. Do not touch
common.py, the stacks, tests, or lambdas/.
Task A: apply spec.appAndPins — add account='328440206208' to BOTH
cdk.Environment calls in app.py, pin constructs== to the exact installed
version in cdk/requirements.txt, and fix the stale "2.259.0" comment(s) at
the file:line spec.appAndPins gives (only those in files you own — if a
stale comment lives in a stack file, note it for the stacks agent, do NOT
edit their file). Every edit here must be logical-ID-neutral (account on
Environment does not change resource logical IDs; verify in your summary).
Task B: README — document cdk/common.py in the CDK/architecture section
(the shared plain-function helpers, the logical-ID-safety rule, the
account/region-derived Bedrock ARN, the new CfnOutputs), and note the
account is now explicit on both stacks. Match existing README style. If the
README documents the stacks' resource inventory, keep it accurate.
Run before returning: ruff check cdk (app.py) and a py_compile of app.py;
you cannot run the full synth without the stacks agent's edits — if you want
to smoke-test, poll cd cdk && npx cdk synth po-ingest -q up to ~10 min, but
a clean app.py parse is sufficient for your scope.
${specBlock}`,
{ label: 'impl:app-pins-readme', model: 'sonnet', phase: 'Implement', schema: IMPL }),
])
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/3 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'cdk/common.py',
'cdk/po_stack.py',
'cdk/wo_stack.py',
'cdk/app.py',
'cdk/requirements.txt',
'README.md',
'tests/',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification — trust nothing self-reported. Run ALL gates,
quoting failures verbatim:
1. pytest -q --no-cov (repo root — all suites green; a synth-only cdk-diff
test may now exist under tests/)
2. ruff check . && ruff format --check .
3. cd cdk && npx cdk synth po-ingest -q && npx cdk synth workorder-ingest -q
(artifact-id selectors, NOT stack_name — constraint 9)
4. ZERO-CODE invariant (constraint 7): git diff ${BASE}...HEAD -- lambdas/
must output NOTHING.
5. NO CONSTRUCT SUBCLASS (constraint 1): grep cdk/common.py for
'class .*Construct' / 'class .*(Construct)' — there must be NONE; every
extracted symbol is a plain 'def'. Report any subclass as a hard FAIL.
6. cdk/common.py imports NO kms/ssm/event_sources (constraint 5) — grep and
confirm.
7. Both cdk.Environment calls in app.py carry account='328440206208';
constructs== is an exact pin (not >=); no "2.259.0" stale comment
remains (grep the cdk/ tree).
8. CfnOutputs: five function ARNs + the consumed table names are present
across the two stacks (grep CfnOutput).
9. git status --porcelain scope check: every modified/added path under
${EXPECTED_SCOPE.join(', ')} (untracked .coverage/.claude/package/
tolerated). No lambdas/ or non-cdk-diff tests/ change.
passed=true only if all green. YOU MAY NOT edit files.`
const diffPrompt = `${PREAMBLE}
You are the ZERO-CDK-DIFF verifier — the load-bearing gate of this phase
(the doc names it the acceptance test). Everything is local synth-vs-synth.
1. From a clean worktree state, synth the BASE templates: check out (via
git worktree add or git stash-free 'git show'-based synth — prefer
'git worktree add /tmp/phase4-base ${BASE}' so HEAD is untouched), then
in that BASE tree cd cdk && npx cdk synth po-ingest -q -o
/tmp/phase4-diff-base && npx cdk synth workorder-ingest -q -o
/tmp/phase4-diff-base.
2. Synth the HEAD templates: in the working tree cd cdk && npx cdk synth
po-ingest -q -o /tmp/phase4-diff-head && npx cdk synth workorder-ingest
-q -o /tmp/phase4-diff-head. (Both by ARTIFACT ID, constraint 9.)
3. Diff the two CloudFormation templates per stack
(/tmp/phase4-diff-base/<stack>.template.json vs
/tmp/phase4-diff-head/<stack>.template.json). Normalize only CDK-tooling
noise (the CDKMetadata Analytics string, asset-hash-derived S3Key values
that were ALSO equal on BASE). Then judge:
- ZERO logical-ID changes (no renamed/removed/added Resources except the
net-new CfnOutputs).
- ZERO alarm property deltas (thresholds, statistics p99/p95, expression
strings, FILL, labels, evaluation_periods, datapoints_to_alarm).
- ZERO IAM deltas: the Bedrock PolicyStatement actions/resources must be
byte-identical after the account/region derivation resolves in-account.
- ZERO DynamoDB table / bucket / DLQ / env / runtime deltas.
- The ONLY allowed net-new is the Outputs block (the five function ARNs
+ consumed table names).
ANY delta outside that allowance = FAIL. Paste the actual per-stack
template diff (or 'identical' with the normalized-noise list).
4. Cross-check with the live tool: cd cdk && npx cdk diff po-ingest ;
npx cdk diff workorder-ingest against the deployed state must also show
only the CfnOutput additions (network permitting; if AWS creds are
read-only-absent, the BASE-vs-HEAD template diff in steps 1-3 is
authoritative).
5. Clean up any /tmp worktrees you created (git worktree remove).
passed=true only if BOTH stacks show zero resource change beyond the
CfnOutput additions.`
const lenses = [
{ key: 'logical-id-safety', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — logical-ID-safety lens (the load-bearing rule,
constraint 1). Try to prove a construct-id or tree-shape change slipped in.
(1) Read cdk/common.py: is EVERY extracted symbol a plain 'def' taking
(scope, id, ...)? Any 'class X(Construct)' / Construct subclass / nested
Construct is an automatic CRITICAL — a subclass reparents the tree and
would attempt REPLACEMENT of the RETAIN-protected purchase-orders /
WorkOrders tables and the named buckets. (2) For every rewritten call site
in both stacks, prove the scope passed is the Stack instance (self), NOT a
new intermediate construct, and the construct id string is IDENTICAL to the
BASE inline id (git diff ${BASE} the id strings). (3) Synth BASE and HEAD
and diff the full Resources logical-ID SET — it must be identical (only
Outputs added). Any renamed/removed logical ID = confirmed CRITICAL.
(4) Confirm the RETAIN removal policies on the tables and the bucket
names/policies are byte-identical post-refactor. confirmed=true only with a
concrete logical-ID delta or a subclass-smell file:line.` },
{ key: 'alarm-variance', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — alarm-variance lens (constraint 3). The shared helpers
must NOT homogenize the deliberate per-function differences. Read
cdk/common.py + every helper call site + the synthesized templates' alarms.
Prove each of these survived EXACTLY: (1) PO email-processor duration alarm
uses p99, wo-email-processor uses p95 — quote both from the synthesized
templates. (2) po-web-ui has ONLY throttles+duration alarms (no errors/dlq
beyond what it had). (3) site_extractor has NO DLQ alarm. (4) wo web_ui has
ZERO alarms — prove the shared helper did NOT silently add any (search the
wo template for any alarm whose dimensions point at the wo web_ui
function). (5) Every bespoke alarm description string is byte-identical to
BASE (diff the AlarmDescription fields). (6) Fallback-rate: PO excludes the
rejected series (rejected_included=False), WO includes it; the two rejected
alarms kept their distinct shapes (PO 6h count-floor, WO 5-min/2-of-6); NO
element-wise MAX anywhere. confirmed=true only with a template-level diff
showing a homogenized or dropped alarm.` },
{ key: 'iam-equivalence', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — IAM-equivalence lens (the cross-family-review surface).
make_bedrock_invoke_statement moved the PolicyStatement construction and
swapped hardcoded 328440206208 for Stack.of(scope).account/.region. Prove
the produced IAM is IDENTICAL. (1) Synth both stacks and extract the Bedrock
PolicyStatement from each template; diff actions and resources against
${BASE}'s synthesized statements — the resolved ARN strings (account +
region substituted) must be byte-identical in-account. (2) Confirm the
resources still enumerate the SAME inference-profile ARN and the SAME
per-region foundation-model ARNs (no region dropped/added, no wildcard
broadening). (3) Confirm effect/conditions/principal attachment unchanged
and the statement is attached to the SAME role. (4) Flag any broadening
(e.g. a Ref/Sub that resolves to a wildcard, or Stack.region producing a
different region than the hardcoded one) as confirmed HIGH. confirmed=true
only with the two synthesized statements diffed.` },
]
let round = 0
let checks = null
let diff = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
() => agent(diffPrompt, { label: `verify:cdk-diff-r${round}`, model: 'opus', phase: 'Verify', schema: DIFF }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
diff = results[1]
confirmed = results.slice(2).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && f.severity !== 'low')
const green = checks && checks.passed && diff && diff.passed
log(`Verify round ${round}: mechanical ${checks && checks.passed ? 'GREEN' : 'RED'}, cdk-diff ${diff && diff.passed ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (green && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit files under: ${EXPECTED_SCOPE.join(', ')}.
Fix EVERY item below minimally; the binding spec and 10 pinned constraints
still hold (a finding that conflicts with a constraint is reported, not
"fixed" — the constraint wins, esp. constraint 1's plain-function /
no-Construct-subclass rule, constraint 3's alarm variance, and constraint 4's
distinct rejected alarms / no element-wise MAX). Re-run the specific failing
gate per fix (the zero-cdk-diff check is authoritative — a fix that
introduces ANY logical-ID or property delta is worse than the finding).
MECHANICAL:\n${checks ? checks.details : '(agent died — rerun all gates)'}
CDK-DIFF:\n${diff ? diff.details : '(agent died — rerun all)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}
${specBlock}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && diff && diff.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
cdkDiff: diff,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
spec,
}
}
// ----------------------------------------------------------------- package
phase('Package')
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent:
1. MANDATORY GPT-4.1 CROSS-FAMILY REVIEW (the doc mandates it for the
make_bedrock_invoke_statement IAM PolicyStatement move, even though
semantics are identical). Capture the Bedrock-statement diff first:
git diff ${BASE}...HEAD -- cdk/common.py cdk/po_stack.py cdk/wo_stack.py
(isolate the make_bedrock_invoke_statement + its two call sites), then
RUN inline:
python3 ~/Documents/repositories/seahaven/security-review/cross_review.py
"Review this CDK IAM PolicyStatement move for breaking changes: the two
inline Bedrock invoke PolicyStatements (hardcoded account 328440206208)
were consolidated into make_bedrock_invoke_statement in cdk/common.py,
deriving the inference-profile + per-region foundation-model ARNs from
Stack.of(scope).account/.region. Confirm the produced actions/resources
are byte-identical in-account and no privilege broadening. <paste diff>"
Record the verdict verbatim in your summary. If cross_review.py is
unavailable, DO NOT block the local commit but flag the review as
OUTSTANDING in your summary (it must be run before merge).
2. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md
and follow it exactly.
3. git add only paths under: ${EXPECTED_SCOPE.join(', ')} and
.claude/workflows/phase-4-cdk-common.js. NOT .coverage, NOT package/,
NOT cdk.out. Verify the staged set with git status.
4. ONE commit; write the message to /tmp/phase4-commit-msg.txt and use
git commit -F /tmp/phase4-commit-msg.txt (backticks in -m get eaten by
zsh). Suggested subject:
"feat: extract cdk/common.py — dedup 379 CDK lines as logical-ID-safe plain helpers (refactor phase 4)"
Body: the plain-function (scope, id, ...) approach and WHY no Construct
subclass (RETAIN-table replacement), the account/region-derived Bedrock
ARN, the zero-cdk-diff acceptance evidence for both stacks, the
account=/constructs pin/stale-comment/CfnOutput net-new additions, and
the cross_review.py verdict one-liner. NO AI attribution / Co-Authored-By
lines.
5. Do NOT push. Return commit sha + shortstat + the cross_review.py verdict
in summary.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
base: BASE,
commit: commit ? commit.summary : 'commit agent died — commit manually',
spec: { commonModule: spec.commonModule, bedrockStatement: spec.bedrockStatement, fallbackRateCalls: spec.fallbackRateCalls, appAndPins: spec.appAndPins, notes: spec.notes },
diffEvidence: diff ? { po: diff.poDiffVerdict, wo: diff.woDiffVerdict } : null,
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
outstandingGates: [
'MANDATORY cross-family GPT-4.1 review (cross_review.py) on the make_bedrock_invoke_statement IAM PolicyStatement move — the Package agent runs it inline and records the verdict; confirm that verdict is clean (or re-run) before merge. If cross_review.py was unavailable at commit time, this review is OUTSTANDING — do not merge without it.',
'/sh-security-review NOT required (pure CDK refactor — no untrusted-input/auth-logic change; the pre-push scanners still run as the unattended backstop)',
'push + PR + gh pr checks green',
'deploy-then-merge with cdk diff zero-change on BOTH stacks as the live acceptance signal: deploy from branch, confirm cdk diff po-ingest / cdk diff workorder-ingest show only the CfnOutput additions, both stacks reach UPDATE_COMPLETE, all alarms still OK, THEN merge (a no-op resource redeploy is itself the verification signal). Drive synth/diff by artifact id workorder-ingest, NOT stack_name WorkorderIngestStack (constraint 9).',
'update the Confluence "AWS Architecture Map" if the new CfnOutputs / account-explicit envs change the documented resource inventory',
],
}

View file

@ -1,692 +0,0 @@
export const meta = {
name: 'phase-5-handler-decomposition',
description: 'Phase 5 of the procurement-ingest refactor (docs/refactor-evaluation.md): decompose both email-processor God-handlers into flat siblings along the seams that already work. PO handler.py splits into handler.py (event loop + auth + routing) / extraction.py (extract_with_claude + prompt import; parse_raw_email already lives in shared/email_parsing.py) / enrichment.py (enrich_parsed, pad_zip — PO-only) / telemetry.py (the EMF ParseMethod emit wrappers) / persistence.py (_write_fields/_merge_update/save_* with save_new_po+save_revision collapsed into one behavior-identical _save_merge). WO splits into ~5 concerns, keeping validate_ai_fallback + the [0-9]+ work_order_id key guard in the handler loop ahead of both saves and _header_date_iso/comment_id determinism with persistence. EXTRACTION_PROMPT moves to prompts.py (re-exported into handler for 4 tests). Lazy cached boto3 accessors in I/O modules; pure modules import no boto3. Requires Phases 0-3 on the base branch (relies on the flat-sibling lambdas/shared/ pattern + the Phase 0/2/3 bundling glob that auto-ships new siblings). Untrusted-input parse/extract/gate/auth-routing code moves, so push is gated on /sh-security-review in the main loop. Committed locally, never pushed.',
phases: [
{ title: 'Setup', detail: 'verify Phases 0+1+2+3 on base (lambdas/shared/ + ses_auth/web_ui_auth/email_parsing/emf present), branch feature/phase-5-handler-decomposition', model: 'haiku' },
{ title: 'Recon', detail: '4 mappers: PO handler seams + save_new_po/save_revision byte-compare, WO handler seams + key-guard/date-determinism, bundling glob + AST test + monkeypatch/loader patch surface, deployed-zip baseline (read-only AWS)' },
{ title: 'Spec', detail: 'serial fable spec: pin per-pipeline module split, _save_merge collapse proof, prompts.py move + re-export, lazy boto3 accessor shape, patch-surface + PO_EXPECTED_TOP_LEVEL_MODULES edits, behavior-pin tests' },
{ title: 'Implement', detail: 'opus: PO siblings + handler; opus: WO siblings + handler; opus: all test plumbing + bundle-test + README — disjoint files', model: 'opus' },
{ title: 'Verify', detail: 'mechanical CHECKS (suite/goldens/ruff/synth/AST/zero cdk+shared+derived diff/ordering greps) + 3 fable lenses (behavior-preservation, boto3-laziness, save-merge-collapse)' },
{ title: 'Fix', detail: 'opus fixer, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'single commit via -F (no push)', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-5-handler-decomposition'
let _args = args
if (typeof _args === 'string') {
try { _args = JSON.parse(_args) } catch (e) { _args = null }
}
const BASE = (_args && _args.base) || 'main'
const CONSTRAINTS = `
PINNED BEHAVIORAL CONSTRAINTS (docs/refactor-evaluation.md Phase 5 — violating any is a build failure):
1. PO SPLIT INTO FLAT SIBLINGS under lambdas/po/email_processor/ (same dir,
flat-landing so bare-name imports keep working — the ses_auth/
template_parser/derive_all pattern):
handler.py -- event loop + fail-closed auth + email_type routing ONLY.
extraction.py -- extract_with_claude + _EMAIL_TAG_RE; imports
EXTRACTION_PROMPT from prompts. parse_raw_email is
ALREADY in shared/email_parsing.py after Phase 3 --
import it, do NOT recreate a PO copy.
enrichment.py -- enrich_parsed + pad_zip (PO-ONLY; WO has NO enrichment
stage). The derived-field block moves here as a PURE
code move.
telemetry.py -- the EMF ParseMethod emit wrappers (_emit_parse_method_metric
and the derived-agreement emit call surface used by
enrichment) -- but see constraint 4: derived_fields.py
and its shadow telemetry BEHAVIOR are untouchable.
persistence.py -- _write_fields / _merge_update / save_*. COLLAPSE the
byte-identical save_new_po/save_revision into ONE
_save_merge, behavior-identical to BOTH originals
INCLUDING the sticky-cancel ConditionExpression guard;
save_cancellation stays its own function.
2. WO SPLIT is ~5 concerns, NOT 7 (WO has no enrichment stage). Its split MUST
keep validate_ai_fallback AND the re.fullmatch(r"[0-9]+", work_order_id) key
guard in the handler loop AHEAD of BOTH save_work_order and save_event
(the guard protects the DynamoDB partition key + the '#'-delimited comment_id
range-key segment), and MUST keep _header_date_iso / comment_id determinism
WITH persistence (the retry-idempotent event_id key). Do not move the guard
below either save; do not fork comment_id determinism from save_event.
3. Move EXTRACTION_PROMPT (the large ~181-line prompt) to prompts.py with
cross-reference headers to derived_fields (the trade/site/fiscal rule tables
have a second authoritative copy there). KEEP a re-export in handler
('from prompts import EXTRACTION_PROMPT') since 4 tests dereference
handler.EXTRACTION_PROMPT. Do the same shape for WO if its prompt moves.
4. BEHAVIOR-PRESERVATION (immovable, pin with tests):
* PO emits ParseMethod=ai_fallback BEFORE the Bedrock call (handler loop:
_emit_parse_method_metric fires before extract_with_claude); the
ai_fallback_rejected emit is the additive second datapoint on rejection.
* WO emits AFTER its gate, with mutually-exclusive ai_fallback /
ai_fallback_rejected (a rejected WO email emits ONLY ai_fallback_rejected).
* Shadow DerivedFieldAgreement telemetry stays ai_fallback-only.
* derived_fields.py and the shadow-telemetry BEHAVIOR are UNTOUCHABLE (the
bake is in progress). Moving enrich_parsed into enrichment.py must be a
PURE code move -- byte-identical logic, ZERO behavior change.
git diff ${BASE}...HEAD -- lambdas/po/email_processor/derived_fields.py
MUST be empty.
5. SIGNATURES PRESERVED: handler(event, context) unchanged (event shape +
return contract identical) on BOTH pipelines; the save_* public contract
unchanged (the _save_merge collapse must be behavior-identical to both
save_new_po and save_revision -- callers route new_po/revision through it).
Goldens UNCHANGED.
6. BUNDLING: the Phase 0 top-level cp glob (cp ./*.py or cp <fn>/*.py) plus the
AST bundle-consistency test already ship & verify new flat siblings
automatically -- but the exact-set pin PO_EXPECTED_TOP_LEVEL_MODULES (and the
WO equivalent) must be updated for the new sibling modules IN THE SAME CHANGE,
keeping the test's teeth (a new sibling that is NOT added to the expected set,
or a commented-out cp, must still fail the AST test).
7. LAZY CACHED boto3 accessors in the I/O modules (extraction.py's bedrock,
persistence.py's dynamodb, handler.py's s3 -- whatever each module actually
uses); PURE modules (enrichment.py logic, prompts.py, telemetry.py if it
makes no AWS call) import NO boto3. Update the
monkeypatch.setattr(handler, "dynamodb", fake) patch surface in the SAME
change -- tests must patch the NEW module's accessor (e.g. the persistence
module), everywhere the tests patch it. Preserve the moto-before-handler
import ordering (_po_parser_support.py:26-33 -- verify current lines) or the
moto-backed suites hit real AWS.
8. ZERO cdk/ diff (git diff ${BASE}...HEAD -- cdk/ must be empty) and ZERO
lambdas/shared/ diff (Phase 3's shared modules are untouchable here).
9. UNTOUCHABLE FILES (git diff ${BASE}...HEAD must be empty for each):
lambdas/po/email_processor/derived_fields.py, both template_parser.py, both
validate_ai_fallback gate modules, lambdas/shared/*, cdk/*, and
lambdas/po/site_extractor/ + both web_ui/ (Phase 5 touches only the two
email processors). No opportunistic refactors -- every handler diff is a
pure move (delete-here/add-there), an import line, or the _save_merge
collapse per the binding spec.
10. AWS access is READ-ONLY (get-function, downloading deployed zips via
presigned URLs). NEVER cdk deploy, never invoke, never mutate.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 5 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push, do NOT run cdk deploy or
touch AWS resources beyond read-only calls).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 5 — Handler
decomposition + lazy boto3 clients".
Work ONLY in the files you are told you own; other agents are concurrently
editing other files in this same working tree.
${CONSTRAINTS}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const SPEC = {
type: 'object',
required: ['poSplit', 'woSplit', 'promptMove', 'botoLaziness', 'saveMergeCollapse', 'testPlumbing', 'behaviorPins', 'notes'],
properties: {
poSplit: { type: 'string', description: 'per PO sibling (handler/extraction/enrichment/telemetry/persistence): exact functions/constants it owns, its imports, and the source file:line ranges moved into it — nothing else may change; state explicitly that parse_raw_email is imported from shared/email_parsing.py and enrich_parsed is a byte-identical move' },
woSplit: { type: 'string', description: 'per WO sibling (~5 concerns): exact ownership, and the pinned proof that validate_ai_fallback + the [0-9]+ key guard stay in the handler loop ahead of both saves and _header_date_iso/comment_id determinism stays with persistence' },
promptMove: { type: 'string', description: 'prompts.py contents (which prompt(s) move), the cross-reference headers to derived_fields, and the exact re-export line kept in each handler so handler.EXTRACTION_PROMPT still resolves (name the 4 tests that dereference it)' },
botoLaziness: { type: 'string', description: 'per module: which boto3 client/resource it needs, the lazy cached accessor shape (module-level None + get_x() cache), which modules import NO boto3, and the exact new monkeypatch target(s) that replace monkeypatch.setattr(handler, "dynamodb", fake) everywhere tests patch it' },
saveMergeCollapse: { type: 'string', description: 'the single _save_merge signature + body, with a line-by-line proof it is behavior-identical to BOTH save_new_po and save_revision including the sticky-cancel ConditionExpression path; how the router calls it for new_po vs revision' },
testPlumbing: { type: 'string', description: 'every test/support file change: loader/_SIBLING_MODULES resolution for the new siblings, _po_parser_support.py/_wo_parser_support.py edits (moto-before-handler ordering preserved), the monkeypatch patch-surface rewrites, PO_EXPECTED_TOP_LEVEL_MODULES + WO equivalent edits with mutation-detection shapes, and the new behavior-pin tests' },
behaviorPins: { type: 'string', description: 'the exact tests/assertions pinning: PO ai_fallback emit precedes the Bedrock invoke; WO emits mutually-exclusive; shadow telemetry ai_fallback-only; _save_merge parity; the WO key-guard ordering' },
notes: { type: 'string' },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
// ------------------------------------------------------------------- setup
phase('Setup')
const setup = await agent(`
In ${REPO}:
1. SEQUENCING GATE — Phases 0, 1, 2 AND 3 must all be on ${BASE}. Phase 5
relies on the flat-sibling lambdas/shared/ pattern + the Phase 0/2/3
bundling glob that auto-ships new siblings, and parse_raw_email now lives in
shared/email_parsing.py after Phase 3. git fetch origin, then pick the base
ref: origin/${BASE} if that remote ref exists, otherwise the local branch
${BASE} (a stacked local-only base is expected and fine). Verify on the base
ref:
(a) Phase 0: tests/test_bundle_consistency.py exists
(git show <baseref>:tests/test_bundle_consistency.py | head -3);
(b) Phase 1: validate_ai_fallback exists in the PO pipeline
(git grep validate_ai_fallback <baseref> -- lambdas/po);
(c) Phase 2: BOTH cdk/po_stack.py and cdk/wo_stack.py on the base ref
contain Code.from_asset("../lambdas") for the email processors
(git show <baseref>:cdk/po_stack.py | grep -n '\\.\\./lambdas', same for
wo_stack.py);
(d) Phase 3: lambdas/shared/ EXISTS on the base ref and the four shared
modules are present — ses_auth.py, web_ui_auth.py, email_parsing.py,
emf.py (git show <baseref>:lambdas/shared/email_parsing.py | head -3 for
each; and confirm parse_raw_email lives in shared/email_parsing.py, NOT
in the PO email_processor dir, on the base ref).
If ANY is missing — ESPECIALLY Phase 3 (no lambdas/shared/ or no
email_parsing.py on base) — STOP with a blocker naming the unmet phase and
do nothing else.
2. Verify clean working tree (untracked .coverage / .claude/ / the local
44 MB lambdas/po/email_processor/package/ dir are fine; any OTHER dirt =
blocker, never stash or discard).
3. git checkout ${BASE}; then git pull --ff-only ONLY if the branch has an
upstream (a local-only base skips the pull — not a blocker); then
git checkout -b ${BRANCH}
4. gh pr list --state open --json number,title,headRefName (overlap check).
Return facts: HEAD sha, per-phase gate evidence (especially the Phase 3
shared/ + email_parsing.py proof), open PRs, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (!setup || (setup.blockers && setup.blockers.length)) {
return { aborted: 'setup blockers', blockers: setup ? setup.blockers : ['setup agent died'], facts: setup ? setup.facts : [] }
}
log(`Branch ${BRANCH} ready off ${BASE}. ${setup.summary}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of the PO email-processor God-handler
(lambdas/po/email_processor/handler.py) and its decomposition seams:
1. Quote with file:line the exact boundaries of every function/constant that
moves and where it lands per constraint 1: EXTRACTION_PROMPT + _EMAIL_TAG_RE
+ extract_with_claude (-> extraction.py/prompts.py); _emit_parse_method_metric
+ _derived_agreement + _emit_derived_agreement_metric (-> telemetry.py, but
note which of these the derived-agreement path needs and whether any lives in
derived_fields.py per constraint 4); enrich_parsed + pad_zip (-> enrichment.py);
_write_fields + _merge_update + save_new_po + save_revision + save_cancellation
(-> persistence.py).
2. save_new_po vs save_revision: diff their BODIES line by line and prove they
are byte-identical modulo the docstring/log string (the D9 collapse target).
Quote both verbatim so the spec can pin _save_merge.
3. The handler loop (handler.py:662-758 currently): quote the exact ordering —
healthcheck early-return, Records loop, s3.get_object, authenticate_inbound_email,
parse_raw_email, try_deterministic_parse, _emit_parse_method_metric (PRE-Bedrock),
extract_with_claude, validate_ai_fallback, ai_fallback_rejected emit + continue,
po_number guard, enrich_parsed, email_type routing to save_*.
4. Which boto3 clients each moving chunk uses (s3, dynamodb resource, bedrock
client) and the current module-level client lines (handler.py:27-29).
5. The current 'from derived_fields import derive_all',
'from ses_auth import authenticate_inbound_email',
'from template_parser import ...' import lines and what parse_raw_email's
import will become (shared/email_parsing.py after Phase 3 -- confirm on the
working tree whether it is already imported from there).
6. Which tests dereference handler.EXTRACTION_PROMPT (grep) and which patch
handler.dynamodb / handler.bedrock / handler.s3.
25-35 precise facts.`,
{ label: 'recon:po-handler', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the WO email-processor God-handler
(lambdas/wo/email_processor/handler.py) and its ~5-concern split:
1. Quote with file:line every function/constant and the concern it belongs to:
EXTRACTION_PROMPT + _EMAIL_TAG_RE + extract_with_bedrock; emit_parse_metric;
save_work_order; _header_date_iso + save_event; the handler loop.
2. The handler loop (handler.py:350-435 currently): quote the exact ordering —
healthcheck, Records loop, s3.get_object, authenticate_inbound_email,
parse_raw_email, try_deterministic_parse, extract_with_bedrock,
validate_ai_fallback, ai_fallback_rejected emit + continue, emit_parse_metric
(POST-gate), the re.fullmatch(r"[0-9]+", work_order_id) key guard,
save_work_order, save_event. CRITICAL per constraint 2: pin that the gate +
the [0-9]+ key guard both precede BOTH saves, and that comment_id determinism
(_header_date_iso + the object-key sha suffix, handler.py:277-347) is coupled
to save_event.
3. Prove WO has NO enrichment stage (no enrich_parsed equivalent) so its split
is 5 concerns not 7.
4. WO's EMF emit is mutually-exclusive (rejected email emits ONLY
ai_fallback_rejected then continues; accepted email emits emit_parse_metric
once AFTER the gate) -- quote the exact lines proving this, contrasted with
PO's pre-Bedrock double-count.
5. boto3 clients WO uses (handler.py:27-29) and which WO tests patch
handler.dynamodb / handler.bedrock and dereference handler.EXTRACTION_PROMPT.
6. _wo_parser_support.py bare-sys.path 'import handler' strategy and how it
will resolve the new siblings.
20-30 precise facts.`,
{ label: 'recon:wo-handler', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the bundling glob, the AST bundle-consistency test, and the
test patch/loader surface this phase must rewire:
1. tests/test_bundle_consistency.py: quote PO_EXPECTED_TOP_LEVEL_MODULES
(currently {"handler","ses_auth","template_parser","derived_fields"} on
pre-Phase-3 — read the value ON THIS BRANCH after Phase 3 has added the
shared modules), whether a WO_EXPECTED_TOP_LEVEL_MODULES (or equivalent)
exists, the _first_party_sibling_imports AST logic, the exact-set assertion,
and every mutation-detection assertion (a commented-out cp / a missing module
in the expected set must fail). Confirm the Phase 0 cp glob form in both
stacks (cp ./*.py or cp <fn>/*.py) auto-ships NEW flat siblings with no cdk
edit needed this phase.
2. tests/conftest.py importlib loader: how it resolves module paths, the
sys.modules save/restore, _SIBLING_MODULES exact current contents (which
bare names it lists per pipeline).
3. lambdas/po/email_processor/tests/_po_parser_support.py: the independent
importlib reimplementation, the load-bearing moto-before-handler import order
(~lines 26-33 — quote it), and how handler + its new siblings get loaded.
4. lambdas/wo/email_processor/tests/_wo_parser_support.py: same.
5. Every place tests do monkeypatch.setattr(handler, "dynamodb"/"bedrock"/"s3",
fake) — grep both pipelines' tests and quote each file:line (this is the
patch surface constraint 7 says must move to the new module accessors).
6. Which tests reference handler.save_new_po / handler.save_revision /
handler.enrich_parsed / handler.extract_with_claude / handler.EXTRACTION_PROMPT
directly (they may need repointing after the split).
20-30 facts.`,
{ label: 'recon:bundling-tests', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only AWS recon (region us-east-1, READ-ONLY): for po-email-processor and
workorder-email-processor (find exact function names via the stacks/aws lambda
list-functions) run aws lambda get-function; download each Code.Location
presigned zip to a scratch dir and produce the COMPLETE top-level file list
(unzip -l) separated into (a) first-party top-level .py, (b) dependency dirs,
(c) anything else; record each function's CodeSha256. This deployed first-party
module set is the baseline the post-split staged bundle is judged against: after
the split the staged bundle's first-party .py set must be exactly the deployed
set MINUS nothing PLUS the new sibling modules (extraction/enrichment/telemetry/
persistence/prompts for PO, the WO equivalents) and with the collapsed
save_new_po/save_revision still resolvable via imports (no NEW top-level module
should vanish that a handler still imports). Confirm parse_raw_email is NOT a
top-level module in either deployed email-processor zip if Phase 3 already
shipped it from shared/ (it lands flat as email_parsing.py). Return the
categorized lists as facts.`,
{ label: 'recon:deployed-baseline', model: 'sonnet', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n')
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
.filter(b => b && !/^\s*(none|n\/a)\b/i.test(b))
log(`Recon complete: ${reconOk.length}/4 mappers, ${reconBlockers.length} advisory notes`)
// Recon "blockers" here are downstream guidance (verify the WO dynamodb
// monkeypatch surface per constraint 7 before persistence.py's split;
// compare the post-split staged bundle against the local pre-Phase-5 bundle,
// NOT the stale pre-Phase-3 deployed AWS zip), NOT stop conditions — main-loop
// verified Phase 3 is on base and the seams move cleanly. The real gates are
// pytest-green (catches any missed monkeypatch surface) and the bundle-parity
// verifier. Fold the notes into the spec context instead of aborting.
const reconAdvisories = reconBlockers.length
? `\n\nRECON ADVISORIES (recon flagged these; they are downstream verification guidance — the impl must update the full dynamodb/bedrock/s3 monkeypatch surface (constraint 7) so no test breaks, and the parity check compares against the local pre-Phase-5 staged bundle not the pre-Phase-3 deployed zip — NOT reasons to stop; pytest-green + bundle-parity are the real gates):\n- ${reconBlockers.join('\n- ')}`
: ''
// -------------------------------------------------------------------- spec
phase('Spec')
const spec = await agent(`${PREAMBLE}
You are the SPEC agent — the single authority that pins every contested
decision BEFORE parallel implementation (parallel leaves cannot see each
other's choices). Using the recon pack below plus your own reads of the actual
files, produce the binding implementation spec:
- poSplit: per constraint 1's five PO siblings — exact functions/constants each
owns, the imports each needs, and the source line ranges moved. State
explicitly: parse_raw_email is imported from shared/email_parsing.py (NOT
recreated); enrich_parsed + its derived-field block move into enrichment.py as
a BYTE-IDENTICAL logic move (constraint 4). Resolve where the derived-agreement
emit surface lives (telemetry.py vs left in enrichment) WITHOUT touching
derived_fields.py.
- woSplit: per constraint 2's ~5 WO concerns — ownership, plus the explicit pin
that validate_ai_fallback + the [0-9]+ key guard stay in the handler loop
AHEAD of both saves and _header_date_iso/comment_id determinism stays coupled
to persistence (save_event). WO has NO enrichment stage.
- promptMove: prompts.py contents (PO prompt for sure; WO prompt too if it moves),
the cross-reference headers to derived_fields, and the EXACT re-export line
each handler keeps so handler.EXTRACTION_PROMPT still resolves (name the 4 PO
tests + any WO tests that dereference it).
- botoLaziness: per module, the lazy cached boto3 accessor shape (module-level
None cache + get_x()); which modules import NO boto3 (pure: enrichment logic,
prompts, telemetry if it emits via print only); and the EXACT new monkeypatch
target(s) replacing monkeypatch.setattr(handler, "dynamodb", fake) at every
test site recon found — pick ONE consistent rule (patch the owning module's
accessor/attribute) and state why. Preserve the moto-before-handler ordering.
- saveMergeCollapse: the single _save_merge signature + body with a line-by-line
proof it is behavior-identical to BOTH save_new_po and save_revision including
the sticky-cancel ConditionExpression path; how the router dispatches new_po
vs revision through it; save_cancellation stays separate.
- testPlumbing: every file, every edit — _SIBLING_MODULES additions for the new
bare names, _po_parser_support.py / _wo_parser_support.py loader edits (moto
ordering preserved), the monkeypatch patch-surface rewrites, repointing any
test that referenced handler.save_new_po/enrich_parsed/etc. to the new module,
PO_EXPECTED_TOP_LEVEL_MODULES + the WO equivalent updated to the new sibling
set (with mutation shapes: a missing sibling in the set, or a commented-out cp,
must FAIL the AST test), and the NEW behavior-pin tests.
- behaviorPins: the exact new tests/assertions pinning constraint 4 & 5 — PO
ai_fallback emit precedes the Bedrock invoke (a throttle test is the cleanest
proof), WO emits mutually-exclusive, shadow telemetry ai_fallback-only,
_save_merge parity to both originals, WO key-guard ordering ahead of both saves.
Recon pack:\n${pack}${reconAdvisories}`,
{ label: 'spec:pin-decomposition', phase: 'Spec', schema: SPEC })
if (!spec) return { aborted: 'spec agent died — rerun workflow', reconBlockers }
const specBlock = `BINDING SPEC (from the spec agent — implement EXACTLY this):\n${JSON.stringify(spec, null, 2)}`
log('Spec pinned: PO split, WO split, prompt move, boto3 laziness, _save_merge collapse, test plumbing, behavior pins')
// --------------------------------------------------------------- implement
phase('Implement')
const impl = await parallel([
() => agent(`${PREAMBLE}
YOU OWN: lambdas/po/email_processor/ EXCEPT its tests/ directory (another agent
owns all test and support files). Do not touch cdk/, lambdas/wo/,
lambdas/shared/, or README.
Task: execute the PO split per spec.poSplit + spec.promptMove +
spec.botoLaziness + spec.saveMergeCollapse:
1. Create prompts.py (EXTRACTION_PROMPT + cross-ref headers); keep the
re-export line in handler.py.
2. Create extraction.py (extract_with_claude + _EMAIL_TAG_RE, lazy bedrock
accessor, imports EXTRACTION_PROMPT from prompts and parse_raw_email from
the shared email_parsing module).
3. Create enrichment.py (enrich_parsed + pad_zip + the derived-field block as a
BYTE-IDENTICAL logic move — derived_fields.py stays untouched; no boto3).
4. Create telemetry.py (the EMF ParseMethod emit wrappers per spec) — pure
(print-only, no boto3) unless spec says otherwise.
5. Create persistence.py (_write_fields, _merge_update, the collapsed _save_merge
behavior-identical to BOTH save_new_po and save_revision incl. the sticky-
cancel guard, save_cancellation; lazy dynamodb accessor).
6. Rewrite handler.py down to the event loop + auth + routing, importing from
the new siblings; keep handler(event, context) signature + the healthcheck
early-return + the PRE-Bedrock ai_fallback emit ordering EXACTLY.
Every change is a pure move / import line / the _save_merge collapse — no
opportunistic refactors (constraint 9). derived_fields.py diff MUST stay empty.
Run before returning: ruff check lambdas/po && ruff format lambdas/po --check,
plus an import smoke: python3 -c with sys.path prepended for lambdas/shared +
the PO function dir, importing every new module + handler (tests are NOT yours
to run — the plumbing agent lands them in parallel).
${specBlock}`,
{ label: 'impl:po-siblings', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: lambdas/wo/email_processor/ EXCEPT its tests/ directory (another agent
owns all test and support files). Do not touch cdk/, lambdas/po/,
lambdas/shared/, or README.
Task: execute the WO ~5-concern split per spec.woSplit + spec.promptMove +
spec.botoLaziness:
1. Create prompts.py if the WO prompt moves (re-export kept in handler).
2. Create extraction.py (extract_with_bedrock + _EMAIL_TAG_RE, lazy bedrock).
3. Create telemetry.py (emit_parse_metric) — pure unless spec says otherwise.
4. Create persistence.py (save_work_order, _header_date_iso, save_event —
comment_id determinism stays coupled here; lazy dynamodb).
5. Rewrite handler.py to the event loop + auth + routing, KEEPING
validate_ai_fallback + the re.fullmatch(r"[0-9]+", work_order_id) key guard
in the loop AHEAD of BOTH saves (constraint 2), the POST-gate mutually-
exclusive emit ordering, the healthcheck early-return, and handler(event,
context) signature EXACTLY.
Pure moves / imports only (constraint 9). No enrichment stage exists — do not
invent one.
Run before returning: ruff check lambdas/wo && ruff format lambdas/wo --check,
plus an import smoke: python3 -c with sys.path prepended for lambdas/shared +
the WO function dir, importing every new module + handler.
${specBlock}`,
{ label: 'impl:wo-siblings', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: all test and support files (tests/ at repo root including
tests/test_bundle_consistency.py and tests/conftest.py,
lambdas/po/email_processor/tests/, lambdas/wo/email_processor/tests/) and
README.md ONLY.
Task A: apply spec.testPlumbing in full — _SIBLING_MODULES additions for the new
bare names, _po_parser_support.py / _wo_parser_support.py loader edits
(PRESERVING the moto-before-handler ordering comment + import), rewrite the
monkeypatch patch surface everywhere tests patch handler.dynamodb/bedrock/s3 to
the new module accessors per spec.botoLaziness, repoint any test that referenced
handler.save_new_po / handler.save_revision / handler.enrich_parsed /
handler.extract_with_claude to the new modules (KEEP handler.EXTRACTION_PROMPT
resolving via the re-export — do NOT repoint those 4 tests), and update
PO_EXPECTED_TOP_LEVEL_MODULES + the WO equivalent to the new sibling set WITHOUT
losing teeth (constraint 6 — a missing sibling in the set or a commented-out cp
must FAIL the AST test; keep existing mutation tests green).
Task B: add the NEW behavior-pin tests per spec.behaviorPins — PO ai_fallback
emit precedes the Bedrock invoke (throttle-based proof), WO emits mutually-
exclusive, shadow telemetry ai_fallback-only, _save_merge parity to both
originals, WO key-guard ordering ahead of both saves.
Task C: README — update the repo-layout / module map for the new PO+WO sibling
modules (the flat-landing import rule already ships them via the Phase 0 glob),
note the save_new_po/save_revision collapse into _save_merge and prompts.py.
Match existing README style; goldens stay UNCHANGED.
Run before returning: pytest -q --no-cov at repo root (must be green against the
OTHER agents' edits — they land in parallel; poll by re-running up to ~10 min
before reporting a blocker) and ruff check on every file you touched.
${specBlock}`,
{ label: 'impl:test-plumbing-readme', model: 'opus', phase: 'Implement', schema: IMPL }),
])
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/3 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'lambdas/po/email_processor/',
'lambdas/wo/email_processor/',
'tests/',
'README.md',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification — trust nothing self-reported. Run ALL gates,
quoting failures verbatim:
1. pytest -q --no-cov (repo root, all three roots green; GOLDENS UNCHANGED —
git diff ${BASE}...HEAD on any golden .json/.eml fixture must be empty).
2. ruff check . && ruff format --check .
3. cd cdk && npx cdk synth po-ingest -q && npx cdk synth workorder-ingest -q
(artifact-id selectors, NOT stack_name). The Docker bundling runs — confirm
each staged email-processor asset now contains the new sibling modules.
4. AST BUNDLE TEST: the test_bundle_consistency.py suite is GREEN and its teeth
hold — on a scratch copy, comment out the cp glob line and drop a sibling
from PO_EXPECTED_TOP_LEVEL_MODULES (and the WO equivalent); EACH mutation
must FAIL the test. Restore.
5. ZERO-DIFF UNTOUCHABLES (each must output NOTHING):
git diff ${BASE}...HEAD -- cdk/
git diff ${BASE}...HEAD -- lambdas/shared/
git diff ${BASE}...HEAD -- lambdas/po/email_processor/derived_fields.py
git diff ${BASE}...HEAD -- lambdas/po/email_processor/template_parser.py
git diff ${BASE}...HEAD -- lambdas/wo/email_processor/template_parser.py
plus both validate_ai_fallback gate modules (resolve their filenames first)
and lambdas/po/site_extractor/ + both web_ui/.
6. SIGNATURES: grep proves handler(event, context) unchanged on both pipelines
and the healthcheck early-return + return {"statusCode": 200, ...} contract
intact. handler.EXTRACTION_PROMPT still resolves (the re-export line is
present in both handlers).
7. ORDERING PINS (grep with line numbers):
- PO: _emit_parse_method_metric / the ai_fallback emit still PRECEDES the
extract_with_claude (Bedrock) invoke in the handler loop.
- WO: the emit stays POST-gate and mutually-exclusive (rejected path emits
ONLY ai_fallback_rejected then continue); the re.fullmatch(r"[0-9]+", ...)
key guard still sits AHEAD of BOTH save_work_order and save_event.
8. BOTO3 LAZINESS: grep proves the pure modules (PO enrichment.py, prompts.py,
telemetry if print-only; WO equivalents) import NO boto3, and the I/O modules
use lazy cached accessors (no module-level eager client that a test cannot
patch). No monkeypatch.setattr(handler, "dynamodb"/... ) remains pointing at
a now-empty handler attribute.
9. git status --porcelain scope check: every modified/added/deleted path under
${EXPECTED_SCOPE.join(', ')} (untracked .coverage/.claude/package/ tolerated).
passed=true only if all green. YOU MAY NOT edit files.`
const lenses = [
{ key: 'behavior-preservation', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — behavior-preservation lens. The decomposition must not
change one observable byte of metric or telemetry behavior. (1) METRIC ORDERING:
read the post-split PO handler loop and prove _emit_parse_method_metric fires
BEFORE extract_with_claude (so a Bedrock-side throttle still records ai_fallback)
and that ai_fallback_rejected is still the additive second datapoint; then the
WO loop and prove the emit is POST-gate and MUTUALLY-EXCLUSIVE (a rejected email
emits ONLY ai_fallback_rejected, an accepted one emits once after the gate).
Construct the emitted EMF envelope from each moved emitter and diff it against
${BASE}'s output — namespace, dimension-set list [["ParseMethod"],["ParseMethod",
"TemplateId"]], property keys/types identical. (2) SHADOW TELEMETRY: prove
DerivedFieldAgreement is still emitted ONLY on the ai_fallback path after
enrich_parsed moved to enrichment.py, and that
git diff ${BASE}...HEAD -- derived_fields.py is EMPTY (the enrich_parsed move is
byte-identical logic). (3) ROUTING: prove email_type routing (cancellation/
revision/new_po for PO; the WO save_work_order + save_event pair) still reaches
the same saves in the same order. confirmed=true only with a concrete
file:line proof or a sample-emission diff.` },
{ key: 'boto3-laziness', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — boto3-laziness lens. (1) PURE MODULES: prove the modules
constraint 7 calls pure (PO enrichment.py, prompts.py, telemetry if print-only;
WO equivalents) import NO boto3 at all — grep + read. A stray 'import boto3' or
eager client in a "pure" module is a finding. (2) LAZY ACCESSORS: in the I/O
modules (extraction bedrock, persistence dynamodb, handler s3) prove the client
is created lazily and cached (module-level None + get_x()), not eagerly at import
— an eager module-level client that tests cannot patch is a finding. (3)
PATCH-SURFACE COMPLETENESS: enumerate EVERY test that previously did
monkeypatch.setattr(handler, "dynamodb"/"bedrock"/"s3", fake) and prove each now
patches the correct NEW module accessor — a test that still patches a now-dead
handler attribute silently exercises the REAL client (or moto-misses) and is a
critical finding. (4) MOTO ORDERING: prove the moto-before-handler import order
survived in _po_parser_support.py and _wo_parser_support.py — reorder-and-run to
confirm the suites would hit real AWS without it. confirmed=true only with
file:line or reproduced-failure evidence.` },
{ key: 'save-merge-collapse', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — save-merge-collapse lens. The one _save_merge that replaces
save_new_po AND save_revision must be byte-behavior-identical to BOTH. (1) Read
_save_merge and BOTH ${BASE} originals; prove line-by-line the field-building,
the _merge_update call, and the sticky-cancel path are identical (incl. the
ConditionExpression "attribute_not_exists(po_status) OR po_status <>
:__cancelled_marker" and the ConditionalCheckFailedException fallback that
re-writes WITHOUT po_status/cancelled_at). (2) Prove the router dispatches
email_type=="revision" and the new_po else-branch both through _save_merge, and
that save_cancellation stayed SEPARATE and unchanged. (3) Construct a scenario
where the collapse could diverge: a revision landing on a Cancelled PO, a new_po
with po_status=None, a new_po carrying a non-cancelled status onto a Cancelled
skeleton — and prove _save_merge behaves exactly as the original would have for
each. (4) Confirm the save_* public contract callers rely on is unchanged and
the goldens still pass. confirmed=true only with a concrete divergence sketch
or a line-by-line equivalence proof.` },
]
let round = 0
let checks = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
confirmed = results.slice(1).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && f.severity !== 'low')
const green = checks && checks.passed
log(`Verify round ${round}: mechanical ${checks && checks.passed ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (green && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit files under: ${EXPECTED_SCOPE.join(', ')}.
Fix EVERY item below minimally; the binding spec and 10 pinned constraints
still hold (a finding that conflicts with a constraint is reported, not "fixed"
— the constraint wins, esp. constraint 4's derived_fields untouchability +
byte-identical enrich move, constraint 5's signature/goldens preservation, and
constraint 8's zero cdk/shared diff). Re-run the specific failing gate/test per
fix.
MECHANICAL:\n${checks ? checks.details : '(agent died — rerun all gates)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}
${specBlock}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
spec,
}
}
// ----------------------------------------------------------------- package
phase('Package')
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent:
1. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md
and follow it exactly.
2. git add only paths under: ${EXPECTED_SCOPE.join(', ')} and
.claude/workflows/phase-5-handler-decomposition.js. NOT .coverage, NOT
package/. Verify the staged set with git status — the new sibling files AND
the deletions of the moved-out code from the handlers MUST be staged.
3. ONE commit; write the message to /tmp/phase5-commit-msg.txt and use
git commit -F /tmp/phase5-commit-msg.txt (backticks in -m get eaten by zsh).
Suggested subject:
"feat: decompose email-processor handlers into flat siblings + lazy boto3 clients (refactor phase 5)"
Body: the PO 5-way split (handler/extraction/enrichment/telemetry/persistence)
with the save_new_po/save_revision -> _save_merge collapse one-liner, the WO
~5-concern split keeping the gate+key-guard in the loop, prompts.py + the
handler.EXTRACTION_PROMPT re-export, the lazy-boto3 + patch-surface move, and
the behavior-preservation pins (PO pre-Bedrock emit, WO mutually-exclusive,
shadow telemetry ai_fallback-only, derived_fields untouched). NO AI
attribution / Co-Authored-By lines.
4. Do NOT push. Return commit sha + shortstat in summary.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
base: BASE,
commit: commit ? commit.summary : 'commit agent died — commit manually',
spec: { poSplit: spec.poSplit, woSplit: spec.woSplit, saveMergeCollapse: spec.saveMergeCollapse, botoLaziness: spec.botoLaziness, notes: spec.notes },
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
outstandingGates: [
'/sh-security-review (MANDATORY before push — the untrusted-input parse/extract/validate-gate/auth-routing code is being MOVED across modules; CLAUDE.md gates untrusted-input handling changes and a move still touches the surface. Run on the committed diff, re-run after any post-review fix)',
'cross-family cross_review.py NOT required (handler(event, context) event/return signatures unchanged and no IAM change — internal module moves only)',
'push + PR + gh pr checks green',
'deploy-then-merge: full suite goldens unchanged, smoke green, ONE live email per pipeline (PO + WO) showing the correct ParseMethod, both the fallback + rejected alarm series behaving; the post-merge CI redeploy is a fresh asset (new sibling files) so confirm the smoke gate + AST bundle test caught any missing-module regression BEFORE merge',
],
}

View file

@ -0,0 +1,585 @@
export const meta = {
name: 'phase-6-site-extractor',
description: 'Phase 6 of the procurement-ingest refactor (docs/refactor-evaluation.md): site_extractor reconciliation — end the three-way site_code contradiction. extract_site_code (lambdas/po/site_extractor/handler.py:34) already prefers record["site_code"] but validates it with the digit-requiring, prefix-anchored SITE_CODE_PATTERN (line 23) which rejects all-letter codes like KLAL (permanent pending-review rows) and accepts overlong junk like DLI6X/SNY55. Fix: validate the direct field with the canonical derived_fields shape + skip-list semantics (fullmatch), import derive_site_code for the fallback path, ship derived_fields.py into the site_extractor asset via the Phase 2/3 bundling mechanism (Phase 3 left site_extractor from_asset UNTOUCHED on purpose), delete the dead [A-Z]{4,5} regex ladder (line 62). derived_fields.py itself stays byte-untouched. Characterization tests pin current behavior first, then flip to fixed in the same PR. Gated on the shadow bake concluding. Validated field can be ai_fallback-sourced, so push is gated on /sh-security-review. Committed locally, never pushed.',
phases: [
{ title: 'Setup', detail: 'HARD-GATE on shadow-bake conclusion (args {bakeConcluded:true}) + Phase 3 on base; branch feature/phase-6-site-extractor', model: 'haiku' },
{ title: 'Recon', detail: '3 mappers: site_extractor handler + the digit-pattern defect, derived_fields canonical shape/skip-list/derive_site_code, post-Phase-3 bundling sites + bundle-consistency test + backfill script (read-only)' },
{ title: 'Spec', detail: 'serial fable spec: pin the handler fix (canonical fullmatch + skip-list), the site_extractor bundling change, backfill repoint-or-delete, the characterization-first-then-flip test plan, bundle/parity rules' },
{ title: 'Implement', detail: 'opus: site_extractor handler + new characterization tests; sonnet: cdk site_extractor from_asset + test_bundle_consistency; opus: backfill_sites.py + README — disjoint files', model: 'opus' },
{ title: 'Verify', detail: 'mechanical gates (suite green, ruff, synth, asset CONTAINS derived_fields.py, tests flipped to fixed, zero derived_fields diff) + 3 fable lenses (shape-parity, bundling, regression)' },
{ title: 'Fix', detail: 'opus fixer, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'single commit via -F (no push)', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-6-site-extractor'
let _args = args
if (typeof _args === 'string') {
try { _args = JSON.parse(_args) } catch (e) { _args = null }
}
const BASE = (_args && _args.base) || 'main'
const CONSTRAINTS = `
PINNED BEHAVIORAL CONSTRAINTS (docs/refactor-evaluation.md Phase 6 + §4 item 6 — violating any is a build failure):
1. THE DEFECT is in lambdas/po/site_extractor/handler.py: extract_site_code
(line 34) already prefers record["site_code"] (line 36), but validates it
with SITE_CODE_PATTERN = re.compile(r"[A-Z]{2,4}\\d{1,2}") (line 23) applied
via .match (prefix-anchored, digit-REQUIRING). That pattern REJECTS
all-letter codes like KLAL — so a KLAL-class site can NEVER self-register
and becomes a permanent pending-review row — and ACCEPTS overlong junk like
DLI6X / SNY55 (prefix match, no end anchor). THE FIX: validate the direct
record["site_code"] field with the CANONICAL derived_fields shape + skip-list
semantics using FULLMATCH (derived_fields.py: _STRICT_CODE_RE at line 50 =
re.compile(r"[A-Z][A-Z0-9]{2,4}") + _is_valid_code at line 97 = "not in _SKIP
and >=2 alphabetic chars", _SKIP at line 69). Import derive_site_code
(derived_fields.py:259) for the FALLBACK path (ship_to / line_items). DELETE
the bespoke regex ladder — the line 62 alternation
r"^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\\b" whose [A-Z]{4,5} arm is dead code
(it can never survive the digit-requiring SITE_CODE_PATTERN.match on the very
next line) — and every other SITE_CODE_PATTERN reference. NET EXPECTED
BEHAVIOR: KLAL is ACCEPTED, DLI6X and SNY55 are REJECTED.
2. SHIP derived_fields.py INTO the site_extractor Lambda asset. Add the cp/root
bundling change for site_extractor's from_asset in cdk/po_stack.py (currently
lines 684-686: Code.from_asset("../lambdas/po/site_extractor",
exclude=["**/__pycache__/**"]) — Phase 3 left this UNTOUCHED on purpose).
THIS phase makes lambdas/po/email_processor/derived_fields.py reachable there,
MIRRORING the Phase 2/3 mechanism the two email processors use (widened
asset root + a scoped cp so the module lands FLAT in /asset-output so the
bare-name import 'from derived_fields import derive_site_code' resolves).
KEEP the __pycache__ / tests excludes intact; do NOT ship tests/ or a stale
package/ into the zip. The spec agent pins the exact root + cp form (Docker
bundling vs a no-Docker local staging — ONE mechanism, exact code).
3. Keep shape/skip validation ON the record["site_code"] field even though it
is ai_fallback-SOURCED (the LLM value is authoritative during/after the bake)
— this is validated authority, NOT blind trust and NOT a regex bypass. Keep
parse_address (line 69) AND the pending-review write flow (write_pending_review
line 223 + the handler else-branch lines 283-290) BYTE-UNCHANGED.
4. Repoint scripts/backfill_sites.py at derive_site_code (it currently does
'from handler import extract_site_code, parse_address, upsert_site' via a
sys.path hack) — OR DELETE it entirely (one-time backfill script; deletion is
acceptable and PREFERRED if simpler). Pick one and state why.
5. CHARACTERIZATION TESTS FIRST. site_extractor is 0% covered today (§2). Write
tests that PIN CURRENT behavior first — INCLUDING the KLAL rejection
divergence and the overlong-junk (DLI6X / SNY55) acceptance, parse_address
variants, and the pending-review fallback — then FLIP the assertions to the
FIXED behavior in the SAME PR, so the change is provably intentional. The
final committed suite asserts FIXED behavior (KLAL accepted, DLI6X/SNY55
rejected); the characterization-of-old-behavior stage is a build step, not a
red suite left behind.
6. derived_fields.py itself stays UNTOUCHED — import it, do NOT edit it.
git diff ${BASE}...HEAD -- lambdas/po/email_processor/derived_fields.py MUST
be empty. The shadow-bake authority model must not shift in this phase.
7. AWS access is READ-ONLY (get-function, downloading the deployed site_extractor
zip via presigned URL to confirm the asset contents). NEVER cdk deploy, never
invoke, never mutate. No push, no PR, no cdk deploy in this workflow.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 6 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push, do NOT run cdk deploy or
touch AWS resources beyond read-only calls).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 6 —
site_extractor reconciliation" and section "§4" item 6 (site_extractor
characterization). The doc wins on any conflict.
Work ONLY in the files you are told you own; other agents are concurrently
editing other files in this same working tree.
${CONSTRAINTS}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const SPEC = {
type: 'object',
required: ['handlerFix', 'bundlingEdit', 'backfillDecision', 'characterizationTests', 'parityRules', 'notes'],
properties: {
handlerFix: { type: 'string', description: 'the exact edit to lambdas/po/site_extractor/handler.py: how the direct record["site_code"] field is validated with the canonical derived_fields fullmatch shape + skip-list, the derive_site_code import + fallback wiring, every line deleted (SITE_CODE_PATTERN def, the line-62 alternation, all SITE_CODE_PATTERN references), and an explicit statement that parse_address + write_pending_review + the pending-review else-branch are byte-unchanged. Prove KLAL accepted, DLI6X/SNY55 rejected under the new logic.' },
bundlingEdit: { type: 'string', description: 'the complete replacement from_asset block for site_extractor in cdk/po_stack.py: widened asset root, the scoped cp form staging both the site_extractor *.py AND lambdas/po/email_processor/derived_fields.py flat, the __pycache__/tests/package excludes, and whether it uses Docker bundling or a no-Docker local staging — ONE mechanism, exact code, mirroring the Phase 2/3 email-processor form.' },
backfillDecision: { type: 'string', description: 'repoint scripts/backfill_sites.py at derive_site_code OR delete it — which, the exact resulting file (or git rm), and why.' },
characterizationTests: { type: 'string', description: 'the new test file(s) under lambdas/po/site_extractor/tests/: how they load the handler (loader idiom), the CURRENT-behavior pins (KLAL rejected, DLI6X/SNY55 accepted, parse_address variants, pending-review fallback) and the FIXED-behavior assertions they flip to, plus any test_bundle_consistency edit if it now pins the site_extractor asset.' },
parityRules: { type: 'string', description: 'how Verify judges the change: the staged site_extractor asset must CONTAIN derived_fields.py (+ the handler and its own siblings, flat) and nothing stray (no tests/, no __pycache__, no package/), asset hash deterministic, and the derived_fields.py diff must be empty vs base.' },
notes: { type: 'string' },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
// ------------------------------------------------------------------- setup
phase('Setup')
// HARD-GATE (a): the shadow bake must have concluded. This workflow cannot
// auto-verify the bake ended, so require an explicit opt-in BEFORE any git work.
if (!_args || _args.bakeConcluded !== true) {
return {
aborted: 'Phase 6 gated on shadow-bake conclusion',
blockers: [
'Phase 6 gated on shadow-bake conclusion — re-run with args {bakeConcluded:true} once the DerivedFieldAgreement bake is signed off and the switchover PR (Python authoritative on ai_fallback) has landed.',
],
}
}
const setup = await agent(`
In ${REPO}:
1. SEQUENCING GATE — Phase 3 (lambdas/shared/ extraction with the widened
../lambdas asset roots) must be on ${BASE}: derived_fields.py must be
shippable into the site_extractor asset via the Phase 2/3 bundling
mechanism, and Phase 3 DELIBERATELY left site_extractor's from_asset
untouched — THIS phase adds the bundling change for site_extractor.
git fetch origin, then pick the base ref: origin/${BASE} if that remote ref
exists, otherwise the local branch ${BASE} (a stacked local-only base is
expected and fine). Verify on the base ref:
(a) Phase 2/3 mechanism present: cdk/po_stack.py on the base ref contains
Code.from_asset("../lambdas") for the PO email processor
(git show <baseref>:cdk/po_stack.py | grep -n '\\.\\./lambdas');
(b) Phase 3 landed: lambdas/shared/ exists on the base ref with a
cp shared/*.py in the email-processor bundling command
(git show <baseref>:cdk/po_stack.py | grep -n 'shared'), and
git show <baseref>:lambdas/shared/ses_auth.py | head -3 succeeds;
(c) site_extractor's from_asset is STILL the untouched narrow form on the
base ref (git show <baseref>:cdk/po_stack.py | grep -n
'../lambdas/po/site_extractor') — this phase is the one that widens it.
If (a) or (b) is missing, STOP with a blocker naming the unmet phase and do
nothing else. If (c) is already widened, note it (someone beat us to it) but
do not treat as a hard blocker.
2. Verify clean working tree (untracked .coverage / .claude/ / the local 44 MB
lambdas/po/email_processor/package/ dir are fine; any OTHER dirt = blocker,
never stash or discard).
3. git checkout ${BASE}; then git pull --ff-only ONLY if the branch has an
upstream (a local-only base skips the pull — not a blocker); then
git checkout -b ${BRANCH}
4. gh pr list --state open --json number,title,headRefName (overlap check).
Return facts: HEAD sha, per-gate evidence, open PRs, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (!setup || (setup.blockers && setup.blockers.length)) {
return { aborted: 'setup blockers', blockers: setup ? setup.blockers : ['setup agent died'], facts: setup ? setup.facts : [] }
}
log(`Branch ${BRANCH} ready off ${BASE}. ${setup.summary}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of the site_extractor DEFECT surface
(lambdas/po/site_extractor/handler.py):
1. Quote SITE_CODE_PATTERN verbatim with file:line and EVERY place it is used
(.match / .search). Note that .match is prefix-anchored and the pattern
requires 1-2 trailing digits.
2. Quote extract_site_code in full with line numbers: the direct-field branch
(record["site_code"]), the ship_to name branches, the line_items branch and
its r"^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\\b" alternation — prove the
[A-Z]{4,5} arm is DEAD (it can never survive the very-next-line
SITE_CODE_PATTERN.match, which requires a digit).
3. Concretely trace KLAL (all letters), DLI6, DLI6X, SNY55 through the CURRENT
code — which return, which fall through — to pin the exact current behavior
the characterization tests must first reproduce.
4. Quote the EXACT boundaries (file:line) of parse_address, write_pending_review
and the handler else-branch that calls write_pending_review — these must stay
byte-unchanged; note everything that must NOT move.
5. Confirm there is NO tests/ dir under lambdas/po/site_extractor today and note
how the email-processor test loaders resolve a bare 'import handler' (the
idiom the new characterization tests will reuse).
20-30 precise facts.`,
{ label: 'recon:site-extractor', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the CANONICAL shape in
lambdas/po/email_processor/derived_fields.py that the fix must adopt:
1. Quote _STRICT_CODE_RE (line ~50), _CODE_TOKEN_RE, _SKIP (line ~69) and
_is_valid_code (line ~97) verbatim with file:line. State exactly what shape
counts as a valid site code (3-5 chars, letter-first, all-letters allowed,
>=2 alphabetic chars, not skip-listed) using FULLMATCH.
2. Quote derive_site_code (line ~259) signature + docstring + return contract
(str | None, NEVER raises) and describe what INPUT it takes (the parsed PO
dict) and the priority-ordered shapes it tries — so the fallback wiring in
the handler passes the right shape of dict.
3. Determine precisely: under the canonical fullmatch shape + _is_valid_code,
is KLAL valid? is DLI6X valid? is SNY55 valid? Show the reasoning token by
token. If the canonical shape does NOT by itself reject DLI6X/SNY55, say so
explicitly and identify what additional check (standalone-token lookaround,
length, _is_valid_code, or applying derive-style resolution to the field)
is required to hit the doc's pinned outcome (KLAL accepted; DLI6X/SNY55
rejected) — this is load-bearing for the spec.
4. List every import derived_fields.py needs at module load (must be
stdlib-only so it can be dropped flat into the site_extractor asset with no
new dependency).
15-25 facts. Flag as a blocker any way the pinned KLAL/DLI6X/SNY55 outcome
cannot be met with derived_fields semantics alone.`,
{ label: 'recon:derived-fields-shape', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the bundling + test + script surface this phase touches:
1. cdk/po_stack.py — quote the site_extractor Function block verbatim with
file:line (currently ~677-694): function_name, runtime, architecture,
handler property, memory/timeout, env, and the from_asset call
(root + exclude) that this phase must widen. Also quote the PO
email-processor from_asset (root ../lambdas + the cp glob + exclude list)
as the Phase 2/3 template to mirror. Confirm no requirements.txt exists for
site_extractor. List every function property that must NOT change.
2. tests/test_bundle_consistency.py — quote PO_EXPECTED_TOP_LEVEL_MODULES and
every test; determine whether ANY assertion currently pins the
site_extractor asset (if none does, note that a new site_extractor pin —
asserting derived_fields.py now ships and a commented-out cp fails — is the
clean way to keep teeth; if one does, quote it).
3. scripts/backfill_sites.py — quote the sys.path hack + the
'from handler import extract_site_code, parse_address, upsert_site' line and
how it uses extract_site_code, so the spec can choose repoint vs delete.
4. Read-only AWS (us-east-1): aws lambda get-function for
po-ingest-site-extractor; download its Code.Location zip to a scratch dir and
produce the COMPLETE unzip -l file list (the baseline the new bundled asset
must cover PLUS derived_fields.py); record CodeSha256.
15-25 facts.`,
{ label: 'recon:bundling-tests-script', model: 'sonnet', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n')
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
log(`Recon complete: ${reconOk.length}/3 mappers, ${reconBlockers.length} blockers`)
if (reconBlockers.length) {
return { aborted: 'recon blockers (likely a shape-parity gap — resolve before implementing)', blockers: reconBlockers, reconPack: pack }
}
// -------------------------------------------------------------------- spec
phase('Spec')
const spec = await agent(`${PREAMBLE}
You are the SPEC agent — the single authority that pins every contested
decision BEFORE parallel implementation (parallel leaves cannot see each
other's choices). Using the recon pack below plus your own reads of the actual
files, produce the binding implementation spec:
- handlerFix: the exact edit to lambdas/po/site_extractor/handler.py.
Validate record["site_code"] with the CANONICAL derived_fields fullmatch
shape + skip-list (constraint 1) so KLAL is ACCEPTED and DLI6X/SNY55 are
REJECTED — if recon found the bare fullmatch shape does not reject
DLI6X/SNY55, pin the exact additional check that does (do NOT reintroduce a
digit requirement). Import derive_site_code from derived_fields for the
fallback path; wire it with the correct input shape. DELETE the
SITE_CODE_PATTERN def, the line-62 alternation, and every SITE_CODE_PATTERN
reference. State EXPLICITLY that parse_address, write_pending_review and the
pending-review else-branch are byte-unchanged.
- bundlingEdit: the complete replacement from_asset block for site_extractor —
widened root + scoped cp staging both site_extractor *.py and
lambdas/po/email_processor/derived_fields.py FLAT into /asset-output, with
the __pycache__/tests/package excludes, mirroring the Phase 2/3
email-processor form. Pick ONE mechanism (Docker bundling vs no-Docker local
staging) and give exact code. Mind that the module must land flat so
'from derived_fields import derive_site_code' resolves at runtime.
- backfillDecision: repoint scripts/backfill_sites.py at derive_site_code OR
git rm it — pick one, give the exact resulting file or the rm, and say why.
- characterizationTests: the new tests under lambdas/po/site_extractor/tests/ —
loader idiom, the CURRENT-behavior pins (KLAL rejected, DLI6X/SNY55 accepted,
parse_address variants, pending-review fallback) that get FLIPPED to
FIXED-behavior assertions in the same change, and any test_bundle_consistency
edit that now pins the site_extractor asset (derived_fields ships;
commented-out cp fails).
- parityRules: exactly how Verify judges the staged asset vs the deployed
baseline — staged site_extractor asset CONTAINS derived_fields.py + handler +
own siblings (flat), nothing stray, asset hash deterministic, and
derived_fields.py diff empty vs ${BASE}.
Recon pack:\n${pack}`,
{ label: 'spec:pin-site-extractor', phase: 'Spec', schema: SPEC })
if (!spec) return { aborted: 'spec agent died — rerun workflow', reconBlockers }
const specBlock = `BINDING SPEC (from the spec agent — implement EXACTLY this):\n${JSON.stringify(spec, null, 2)}`
log('Spec pinned: handler fix, site_extractor bundling, backfill decision, characterization tests, parity rules')
// --------------------------------------------------------------- implement
phase('Implement')
const impl = await parallel([
() => agent(`${PREAMBLE}
YOU OWN: lambdas/po/site_extractor/ ONLY (the handler AND a new tests/
subdir). Do not touch cdk/, scripts/, tests/ at repo root, or README.
Task: execute spec.handlerFix + spec.characterizationTests.
1. Edit handler.py per spec.handlerFix: validate record["site_code"] with the
canonical derived_fields fullmatch shape + skip-list, import
derive_site_code for the fallback path, DELETE SITE_CODE_PATTERN and the
line-62 [A-Z]{4,5} alternation and every SITE_CODE_PATTERN reference. Keep
parse_address, write_pending_review and the pending-review else-branch
BYTE-UNCHANGED (constraint 3). Do NOT edit derived_fields.py (constraint 6).
2. Write the characterization tests under lambdas/po/site_extractor/tests/:
FIRST pin CURRENT behavior (KLAL rejected, DLI6X/SNY55 accepted,
parse_address variants, pending-review fallback), THEN flip the assertions
to the FIXED behavior (KLAL accepted, DLI6X/SNY55 rejected) in this same
change — the COMMITTED suite asserts FIXED behavior (constraint 5). The
tests must import derived_fields via the same flat/loader idiom the asset
uses so they exercise the real derive_site_code.
Run before returning: ruff check lambdas/po/site_extractor && ruff format
lambdas/po/site_extractor --check, and pytest -q --no-cov on your new suite
(with derived_fields.py resolvable on sys.path the way the asset ships it —
stub AWS/boto env as needed). git diff ${BASE}...HEAD --
lambdas/po/email_processor/derived_fields.py MUST be empty.
${specBlock}`,
{ label: 'impl:site-extractor', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: cdk/po_stack.py (the site_extractor from_asset region ONLY — alarms,
IAM, tables, env, and the email-processor/web_ui assets are all off-limits) and
tests/test_bundle_consistency.py.
Task: apply spec.bundlingEdit (widen the site_extractor asset root + scoped cp
staging derived_fields.py flat, mirroring the Phase 2/3 email-processor form,
keeping the __pycache__/tests/package excludes) and, per
spec.characterizationTests / spec.parityRules, add or update the
site_extractor pin in test_bundle_consistency.py so a commented-out or narrowed
derived_fields cp FAILS the test (keep existing tests green, don't lose teeth).
Touch nothing else; every other function property and every other asset stays
byte-identical.
Run before returning: ruff check cdk tests/test_bundle_consistency.py && cd cdk
&& npx cdk synth po-ingest -q -o /tmp/phase6-synth (artifact-id selector, NOT
stack_name). Confirm the staged site_extractor asset CONTAINS derived_fields.py
flat alongside handler.py and NO tests/__pycache__/package; note the asset hash.
If the handler edits have not landed yet the synth still works (asset is just
files) — but if the new tests/ dir would be staged, prove the exclude drops it.
${specBlock}`,
{ label: 'impl:cdk-bundle-test', model: 'sonnet', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: scripts/backfill_sites.py and README.md ONLY.
Task A: apply spec.backfillDecision — either repoint scripts/backfill_sites.py
at derive_site_code (importing from derived_fields, not the handler's deleted
extractor) OR git rm it. Do exactly what the spec pinned.
Task B: README — update the site_code / site_extractor section to reflect the
single reconciled definition (derived_fields canonical shape is now
authoritative for site_extractor too; KLAL-class all-letter codes self-register;
derived_fields.py ships into the site_extractor asset via the same bundling cp).
Fix any stale "three inconsistent site_code definitions" wording (finding X2).
Match existing README style.
Run before returning: ruff check on any script you kept, and pytest -q --no-cov
at repo root (must be green against the OTHER agents' edits — they land in
parallel; poll by re-running up to ~10 min before reporting a blocker).
${specBlock}`,
{ label: 'impl:backfill-readme', model: 'opus', phase: 'Implement', schema: IMPL }),
])
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/3 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'lambdas/po/site_extractor/',
'cdk/po_stack.py',
'scripts/backfill_sites.py',
'tests/test_bundle_consistency.py',
'README.md',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification — trust nothing self-reported. Run ALL gates,
quoting failures verbatim:
1. pytest -q --no-cov (repo root, all roots green — including the NEW
site_extractor suite). Confirm the new suite actually exists and is
collected (it did not exist before this phase).
2. ruff check . && ruff format --check .
3. cd cdk && npx cdk synth po-ingest -q (artifact-id selector, NOT stack_name).
4. ASSET CONTENTS: locate the staged site_extractor asset under the synth -o
dir and prove it CONTAINS derived_fields.py flat alongside handler.py and
the site_extractor's own siblings, and contains NO tests/, NO __pycache__,
NO package/ (constraint 2). List the asset's top-level files.
5. CHARACTERIZATION: confirm the committed site_extractor tests now assert the
FIXED behavior — KLAL ACCEPTED, DLI6X and SNY55 REJECTED — and that a
current-behavior-of-old-code pin was not left red (constraint 5).
6. UNTOUCHABLE: git diff ${BASE}...HEAD --
lambdas/po/email_processor/derived_fields.py MUST output NOTHING
(constraint 6). Also confirm parse_address + write_pending_review + the
pending-review else-branch in site_extractor/handler.py are byte-unchanged
vs ${BASE} (git diff the file and check only extract_site_code /
SITE_CODE_PATTERN / imports changed).
7. DELETION: SITE_CODE_PATTERN is gone from site_extractor/handler.py (grep
returns nothing) and no path still uses the deleted digit-requiring regex or
the line-62 [A-Z]{4,5} alternation.
8. git status --porcelain scope check: every modified/added/deleted path under
${EXPECTED_SCOPE.join(', ')} (untracked .coverage/.claude/package/
tolerated). If backfill_sites.py was deleted, the deletion is staged/tracked.
passed=true only if all green. YOU MAY NOT edit files.`
const lenses = [
{ key: 'shape-parity', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — shape-parity lens. Prove extract_site_code now matches
derive_site_code's canonical shape + skip-list. (1) Byte-read the fixed
extract_site_code and derived_fields' _STRICT_CODE_RE / _is_valid_code / _SKIP;
construct the token set {KLAL, DLI6, DLI6X, SNY55, RME (skip-listed), B187
(<2 alpha)} and trace each through the fixed direct-field validation AND the
derive_site_code fallback — KLAL MUST be accepted, DLI6X and SNY55 MUST be
rejected, skip-listed and sub-2-alpha tokens rejected. (2) Prove NO path still
uses the deleted digit-requiring SITE_CODE_PATTERN or the [A-Z]{4,5} dead arm
(grep the whole handler). (3) Attack the fallback wiring: does derive_site_code
receive the dict shape it expects, and does it NEVER raise (a stream record with
missing/empty fields must not throw)? confirmed=true only with a concrete
token-trace or file:line proof.` },
{ key: 'bundling', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — bundling lens. Prove derived_fields.py actually ships in
the site_extractor zip and the excludes stay intact. (1) cd cdk && npx cdk
synth po-ingest -q -o /tmp/phase6-bundle-a, then again into
/tmp/phase6-bundle-b; locate the staged site_extractor asset in each and prove
derived_fields.py is present flat next to handler.py, that the runtime import
'from derived_fields import derive_site_code' resolves with that dir alone on
sys.path (python3 -c with env stubs), and that the two synths produce IDENTICAL
asset hashes (determinism). (2) Drop a throwaway __pycache__/junk.pyc and a
tests/scratch_test.py under the widened root region (delete them afterwards),
re-synth, and confirm the excludes keep the asset hash unchanged and neither
file appears. (3) Confirm NO other asset (email processors, web_ui) changed
hash as a side effect of the root widening. confirmed=true only with evidence.` },
{ key: 'regression', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — regression lens. The fix must be surgical. (1) git diff
${BASE}...HEAD -- lambdas/po/site_extractor/handler.py — prove parse_address,
write_pending_review, upsert_site, load_address_cache, lookup_by_address,
deserialize_image and the handler() event loop / pending-review else-branch are
BYTE-UNCHANGED; the ONLY deltas are inside extract_site_code + its imports +
the deleted SITE_CODE_PATTERN. (2) Prove derived_fields.py is byte-unchanged
vs ${BASE} (constraint 6) — the shadow-bake authority model did not shift.
(3) Confirm the ai_fallback-sourced record["site_code"] is still VALIDATED
(shape + skip-list), not blindly trusted and not regex-bypassed (constraint 3).
(4) If backfill_sites.py was kept, prove it no longer imports the deleted
extractor; if deleted, prove nothing else imports it. confirmed=true only with
file:line or diff evidence.` },
]
let round = 0
let checks = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
confirmed = results.slice(1).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && f.severity !== 'low')
const green = checks && checks.passed
log(`Verify round ${round}: mechanical ${checks && checks.passed ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (green && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit files under: ${EXPECTED_SCOPE.join(', ')}.
Fix EVERY item below minimally; the binding spec and 7 pinned constraints still
hold (a finding that conflicts with a constraint is reported, not "fixed" — the
constraint wins, esp. constraint 6's derived_fields untouchability and
constraint 3's validated-not-blind-trust rule). Re-run the specific failing
gate/test per fix.
MECHANICAL:\n${checks ? checks.details : '(agent died — rerun all gates)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}
${specBlock}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
spec,
}
}
// ----------------------------------------------------------------- package
phase('Package')
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent:
1. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md
and follow it exactly.
2. git add only paths under: ${EXPECTED_SCOPE.join(', ')} and
.claude/workflows/phase-6-site-extractor.js. NOT .coverage, NOT package/.
Verify the staged set with git status — if backfill_sites.py was deleted,
its deletion MUST be staged too.
3. ONE commit; write the message to /tmp/phase6-commit-msg.txt and use
git commit -F /tmp/phase6-commit-msg.txt (backticks in -m get eaten by zsh).
Suggested subject:
"feat: reconcile site_extractor site_code with derived_fields canonical shape (refactor phase 6)"
Body: the digit-pattern defect (KLAL never self-registered, DLI6X/SNY55
false-accepted), the canonical fullmatch + skip-list fix + derive_site_code
fallback, the derived_fields.py bundling cp into the site_extractor asset
(Phase 2/3 mechanism), the characterization-first-then-flip test approach,
the backfill repoint/delete decision, and that derived_fields.py is
byte-unchanged. NO AI attribution / Co-Authored-By lines.
4. Do NOT push. Return commit sha + shortstat in summary.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
base: BASE,
commit: commit ? commit.summary : 'commit agent died — commit manually',
spec: { handlerFix: spec.handlerFix, bundlingEdit: spec.bundlingEdit, backfillDecision: spec.backfillDecision, notes: spec.notes },
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
outstandingGates: [
'/sh-security-review RECOMMENDED before push — the validated record["site_code"] field can be ai_fallback-sourced (attacker-influenced via a DKIM-passing injected body), so the new shape/skip validation is untrusted-input handling; run it on the committed diff and flag it.',
'cross-family cross_review.py NOT required (no IAM/policy change, no Lambda handler signature/event/return-contract change — internal validation logic + a bundling cp only).',
'push + PR + gh pr checks green.',
'deploy-then-merge: deploy from branch, smoke green, and confirm the LIVE success signal = the pending-site-review write rate DROPS (KLAL-class all-letter codes now self-register instead of falling to pending review), plus one real DynamoDB-stream event processed clean, THEN merge.',
'GOTCHA: npx cdk synth po-ingest uses the ARTIFACT-ID selector (po-ingest), not the stack_name — using the wrong selector silently synths nothing or the wrong stack.',
],
}

View file

@ -1,661 +0,0 @@
export const meta = {
name: 'phase-7-ops-recovery',
description: 'Phase 7 of the procurement-ingest refactor (docs/refactor-evaluation.md): ops/recovery + dependency hygiene. Generalize scripts/reprocess.py (--pipeline po|wo, --key/--prefix/--since; TARGETED replay primary, full-prefix DEMOTED behind --all with documented caveats) + a synthetic-event-shape contract test pinning the raw-key/no-URL-decode S3 event. New docs/runbook-dlq-recovery.md for the no-redrive async-destination DLQ (receive->key->re-invoke->verify->purge; 14d DLQ / 90d S3 windows; sender-auth + ai_fallback_rejected drops never reach the DLQ), linked from README alarms. Dependency hygiene: drop vendored boto3 from both email-processor requirements (handbook empty-with-comment form; runtime copy suffices per lambda-template.md), simplify bundling to cp-only where the vendored dep is gone, exact-pin moto==, pin-then-add dependabot entries for /tests + po/web_ui + po/site_extractor, reduce wo/web_ui dead manifest to empty-with-comment. Optional local-only rm of the untracked 44 MB package/ dir. Gates on Phase 2 merged (../lambdas asset root). Ops tooling + dep hygiene, no auth/untrusted-input/IAM change. Committed locally, never pushed.',
phases: [
{ title: 'Setup', detail: 'verify Phase 2 on base (Code.from_asset("../lambdas")), branch feature/phase-7-ops-recovery', model: 'haiku' },
{ title: 'Recon', detail: '4 mappers: reprocess.py + event shape, dependency/manifest/dependabot state, post-Phase-2 bundling + package/ dir, DLQ/alarm/S3-lifecycle names (read-only AWS)' },
{ title: 'Spec', detail: 'serial fable spec: reprocess CLI + caveats, runbook contents, requirements/moto/dependabot edits, cp-only bundling form, package/ cleanup, contract test, verify rules' },
{ title: 'Implement', detail: 'opus: scripts/ + reprocess contract test; sonnet: requirements + dependabot + cdk bundling; opus: runbook + README — disjoint files', model: 'opus' },
{ title: 'Verify', detail: 'mechanical gates + 3 fable lenses (bundling-integrity, reprocess-safety, dependency)', model: 'sonnet' },
{ title: 'Fix', detail: 'opus fixer, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'single commit via -F (no push); optional local rm of package/ reported, not committed', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-7-ops-recovery'
let _args = args
if (typeof _args === 'string') {
try { _args = JSON.parse(_args) } catch (e) { _args = null }
}
const BASE = (_args && _args.base) || 'main'
const CONSTRAINTS = `
PINNED BEHAVIORAL CONSTRAINTS (docs/refactor-evaluation.md Phase 7 — violating any is a build failure):
1. GENERALIZE scripts/reprocess.py: add --pipeline po|wo (selects the
function name + bucket per pipeline — po-email-processor /
po-ingest-emails-<acct>, workorder-email-processor / its bucket), plus
--key (single object), --prefix, and --since (time filter on
LastModified). TARGETED replay (--key / --prefix / --since) is the
PRIMARY, default mode. FULL-PREFIX replay is DEMOTED behind an explicit
--all flag. The --all caveats MUST be documented BOTH in --help text and
in code comments: Event-type (async) invocation is CONCURRENT so sorting
does NOT serialize; switch to RequestResponse if order matters; metrics
get double-counted on replay; Bedrock is re-billed; out-of-order replay
can REGRESS already-merged fields. Keep the existing dry-run-by-default /
--execute safety (targeted replay stays dry-run unless --execute).
2. SYNTHETIC-EVENT-SHAPE CONTRACT TEST (§4.7): pin the exact S3 event shape
reprocess emits — Records[0].s3.bucket.name + Records[0].s3.object.key —
and assert the key is the RAW object key with NO URL-decoding (a real S3
notification URL-encodes the key; reprocess builds from the raw
list_objects_v2 key, and the handler is what decodes — replaying a
pre-decoded key would double-decode). Test lives under scripts' tests or
tests/ and runs in the standard pytest collection.
3. NEW docs/runbook-dlq-recovery.md: the async on-failure destination DLQ
has NO console redrive-to-source. Documented procedure: receive-message
-> extract the S3 key from the event body -> targeted re-invoke (via the
generalized reprocess.py --key) -> verify the write -> purge the message.
State the recovery WINDOWS explicitly: 14-day DLQ breadcrumb retention,
90-day raw-email S3 retention (the inbound/ lifecycle rule OVERRIDES the
table RETAIN policy — S3 is the real replay floor). Document that
sender-auth rejections AND ai_fallback_rejected drops INTENTIONALLY never
reach the DLQ (they are fail-closed SKIPS, not errors — no retry, no DLQ
message). Link the runbook from the README alarms section.
4. DEPENDENCY HYGIENE — DROP vendored boto3 from BOTH email-processor
requirements.txt: replace 'boto3>=1.43.47' with the handbook
empty-with-comment form (a comment explaining the Lambda runtime provides
boto3 per lambda-template.md — only-boto3 functions correctly use the
runtime copy; no pinned third-party dep remains). Same empty-with-comment
reduction for wo/web_ui/requirements.txt (a dead manifest today).
5. SIMPLIFY BUNDLING TO CP-ONLY where the vendored dep is gone: once the
email-processor requirements are empty-with-comment, 'pip install -r ... '
installs nothing, so the pip step is removable and the command becomes
cp-only. This is ONLY safe BECAUSE there are no third-party binary deps
left. Do NOT regress: the Phase 2 exclude lists
(['**/__pycache__/**','**/tests/**','**/package/**']) STAY; the widened
'../lambdas' asset root STAYS; if a Phase 3 'cp shared/*.py' line is
present on the base it STAYS. The '--platform manylinux2014_aarch64
--only-binary=:all:' pin is only meaningful while a pip install runs — its
accidental removal caused the PR #34 outage, so removing the WHOLE pip
step (deliberately, because nothing is installed) is the only acceptable
way it disappears; you may NOT keep a pip install while dropping the pin.
If recon finds ANY third-party dep still required by an email processor,
cp-only is a blocker — keep the pinned pip step.
6. EXACT-PIN moto: change tests/requirements.txt 'moto>=5.0.0' to
'moto==<resolved-version>' (resolve the actually-installed/current 5.x
version; floor pins make the dependabot entry a no-op).
7. DEPENDABOT: add pip entries for '/tests', '/lambdas/po/web_ui', and
'/lambdas/po/site_extractor' — but PIN THOSE MANIFESTS FIRST. po/web_ui
and po/site_extractor have NO requirements.txt today, so a dependabot
entry pointed at them is a no-op until a PINNED manifest exists; create
each manifest (exact-pinned boto3, or empty-with-comment ONLY if the
dependabot entry would then be pointless — a dependabot entry needs at
least one pinned dep to act on). Match the existing dependabot.yml entry
shape (weekly, minor-and-patch group). tests/ gets its entry once moto is
exact-pinned (constraint 6).
8. DELETE the untracked 44 MB lambdas/po/email_processor/package/ dir as an
OPTIONAL local-cleanup step: it is UNTRACKED / local-only, so this is a
filesystem 'rm -rf', NOT a 'git rm' (there is nothing tracked to commit).
With Phase 2's '**/package/**' exclude already deployed, the deletion is
asset-hash-NEUTRAL. The Package agent PERFORMS and REPORTS it; it is NOT
part of the commit.
9. UNTOUCHABLE: no handler / lambda runtime code change under lambdas/*/
email_processor/*.py (this is dep + ops + bundling only). No alarm, IAM,
table, env, runtime, memory, timeout, or logical-ID delta on either stack
— the ONLY cdk delta permitted is the bundling command string on the two
email processors (+ any resulting asset-hash / CDK metadata change). The
dropped-vendored-boto3 functions MUST still 'import boto3' at runtime from
the Lambda runtime copy.
10. AWS access is READ-ONLY (get-function, get-function-event-invoke-config,
SQS get-queue-attributes, S3 get-bucket-lifecycle-configuration,
describe-alarms). NEVER cdk deploy, never invoke, never mutate, never
purge a real queue.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 7 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push, do NOT run cdk deploy or
touch AWS resources beyond read-only calls).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 7 — Ops/recovery + dependency hygiene".
Work ONLY in the files you are told you own; other agents are concurrently
editing other files in this same working tree.
${CONSTRAINTS}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const SPEC = {
type: 'object',
required: ['reprocessCli', 'contractTest', 'runbook', 'dependencyEdits', 'bundlingEdits', 'packageCleanup', 'verifyRules', 'notes'],
properties: {
reprocessCli: { type: 'string', description: 'the complete generalized scripts/reprocess.py design: argparse surface (--pipeline po|wo, --key/--prefix/--since, --all, --execute), per-pipeline function-name/bucket resolution, which modes are dry-run-by-default, and the EXACT --help + code-comment caveat text for --all (concurrency/no-serialize, RequestResponse-if-order-matters, double-count, Bedrock re-bill, out-of-order regression)' },
contractTest: { type: 'string', description: 'the synthetic-event-shape contract test: file path, what it asserts about Records[0].s3.bucket.name / object.key, and the exact raw-key / no-URL-decoding assertion (encoded vs raw key example)' },
runbook: { type: 'string', description: 'the docs/runbook-dlq-recovery.md outline: the no-redrive procedure steps (receive->key->targeted re-invoke via reprocess --key->verify->purge), the 14d-DLQ / 90d-S3 windows with the lifecycle-overrides-RETAIN note, the intentional sender-auth + ai_fallback_rejected non-DLQ statement, and the exact README alarms-section link edit' },
dependencyEdits: { type: 'string', description: 'per requirements.txt: exact new contents (both email-processor empty-with-comment, wo/web_ui empty-with-comment, po/web_ui + po/site_extractor new pinned manifests, tests/ moto==<version>); the resolved moto version; the exact dependabot.yml additions matching the existing entry shape' },
bundlingEdits: { type: 'string', description: 'both email-processor bundling command strings BEFORE and AFTER, verbatim: the cp-only result, proof that the pip step is safely removable (no third-party dep remains), and confirmation the exclude lists + widened root + any Phase-3 shared cp line are preserved' },
packageCleanup: { type: 'string', description: 'the package/ dir cleanup: the exact rm command, why it is a filesystem rm not git rm (untracked), and why it is asset-hash-neutral given the Phase 2 exclude — this is a REPORTED local step, not a commit' },
verifyRules: { type: 'string', description: 'how Verify judges success: which pytest tests must pass (incl. the new contract test), ruff scope now including scripts/, both cdk synth expectations, the dependabot yaml-parse check, requirements resolve/import check, and the runtime import-boto3 proof for the cp-only assets' },
notes: { type: 'string' },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
// ------------------------------------------------------------------- setup
phase('Setup')
const setup = await agent(`
In ${REPO}:
1. SEQUENCING GATE — Phase 2 must be on ${BASE}. Phase 7 "can run parallel
to 4-6" (largely independent), BUT the cp-only bundling simplification and
the package/ cleanup touch the same cdk bundling that Phases 2/3
established, so this phase must NOT race Phase 2. git fetch origin, then
pick the base ref: origin/${BASE} if that remote ref exists, otherwise the
local branch ${BASE} (a stacked local-only base is expected and fine).
Verify on the base ref that BOTH cdk/po_stack.py and cdk/wo_stack.py
contain Code.from_asset("../lambdas") for the email processors
(git show <baseref>:cdk/po_stack.py | grep -n '\\.\\./lambdas', same for
wo_stack.py). If either is missing, STOP with a blocker naming Phase 2 as
unmet and do nothing else.
NOTE (report, do not block): Phase 7 is otherwise independent of Phases
4-6. If Phase 4 (cdk/common.py dedup) lands on ${BASE} FIRST, the
stack-touching bundling edits here must be rebased onto the moved bundling
code — flag that in your facts so the implement agents know whether the
bundling lives inline or in a common helper.
2. Verify clean working tree (untracked .coverage / .claude/ / the local
44 MB lambdas/po/email_processor/package/ dir are fine; any OTHER dirt =
blocker, never stash or discard).
3. git checkout ${BASE}; then git pull --ff-only ONLY if the branch has an
upstream (a local-only base skips the pull — not a blocker); then
git checkout -b ${BRANCH}
4. gh pr list --state open --json number,title,headRefName (overlap check —
especially any in-flight Phase 2/3/4 stack PR).
Return facts: HEAD sha, Phase 2 gate evidence, whether Phase 4 has landed
(bundling inline vs helper), open PRs, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (!setup || (setup.blockers && setup.blockers.length)) {
return { aborted: 'setup blockers', blockers: setup ? setup.blockers : ['setup agent died'], facts: setup ? setup.facts : [] }
}
log(`Branch ${BRANCH} ready off ${BASE}. ${setup.summary}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of scripts/reprocess.py and the event shape it must pin:
1. Quote scripts/reprocess.py verbatim with file:line — the argparse surface
today (--execute only), FUNCTION_NAME, get_bucket_name (po-only,
po-ingest-emails-<acct>), list_inbound_keys (paginator, Prefix='inbound/'),
build_s3_event (the exact Records/s3/bucket.name/object.key shape), the
Event (async) InvocationType.
2. Find the WO analogues so --pipeline wo can resolve: the workorder
email-processor function name and its email bucket naming (grep the cdk
stacks / function_name / bucket_name). State them exactly.
3. Determine how the email-processor handler reads the key from an S3 event:
does it URL-decode Records[].s3.object.key (urllib.parse.unquote_plus)?
Quote the handler line. This decides the contract test's no-double-decode
assertion (reprocess must emit the RAW key so the handler decode path is
exercised exactly once — a pre-decoded key double-decodes '+'/'%20').
4. Is there ANY existing test that touches scripts/reprocess.py? (grep tests/
and scripts/). Where should the new synthetic-event contract test live so
it is collected by the standard pytest run (repo-root pytest collects
tests/ and both email_processor tests/ roots)? Note whether scripts/ is on
sys.path for import.
15-25 precise facts.`,
{ label: 'recon:reprocess', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the dependency / manifest / dependabot state:
1. Quote every requirements.txt under lambdas/ and tests/ verbatim with path:
lambdas/po/email_processor (boto3>=1.43.47), lambdas/wo/email_processor
(boto3>=1.43.47), lambdas/wo/web_ui (boto3>=1.43.47 — dead manifest),
tests/ (moto>=5.0.0), cdk/. Confirm po/web_ui and po/site_extractor have
NO requirements.txt (they must be CREATED + pinned before a dependabot
entry is not a no-op).
2. Resolve the moto version to exact-pin: what 5.x is actually installed /
currently resolves (pip show moto or the lockfile / venv)? State the exact
version string to pin.
3. Quote .github/dependabot.yml verbatim: the existing pip entries (/cdk,
/lambdas/po/email_processor, /lambdas/wo/email_processor, /lambdas/wo/web_ui)
and the github-actions entry — the exact shape (weekly interval,
minor-and-patch group) the three NEW entries must match.
4. Confirm via lambda-template.md (handbook) the exact empty-with-comment
form for only-boto3 functions (the Lambda runtime provides boto3) — quote
the canonical comment wording so the edits match the handbook.
5. Does anything under lambdas/*/email_processor actually import a THIRD-PARTY
package other than boto3 (grep imports; anthropic/requests/etc.)? If yes,
cp-only is unsafe — that is a blocker for constraint 5.
15-25 facts.`,
{ label: 'recon:dependencies', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the post-Phase-2 CDK bundling + the package/ dir:
1. Quote BOTH email-processor Code.from_asset blocks verbatim with current
file:line (cdk/po_stack.py ~241, cdk/wo_stack.py ~240): asset root, the
FULL exclude list, the entire bundling command list — the pip install
'--platform manylinux2014_aarch64 --only-binary=:all: -r
<pipeline>/email_processor/requirements.txt -t /asset-output' step, the
'&&', the 'cp <pipeline>/email_processor/*.py /asset-output/' step, and
whether a Phase-3 'cp shared/*.py /asset-output/' line is already present
on this branch's base. Note the explanatory NOTE comment in po_stack.
2. State the exact cp-only rewrite for each (drop the pip step entirely,
keep every cp line + the widened root + the exclude list). Confirm the
pip step is the ONLY thing removed.
3. The three plain from_asset calls (po/web_ui, po/site_extractor,
wo/web_ui) — quote them; they are NOT bundling-simplified here (no pip
step), just confirm they are untouched by this phase.
4. lambdas/po/email_processor/package/ — du -sh it, confirm it is UNTRACKED
(git status / git ls-files must not list it), and confirm the Phase 2
'**/package/**' exclude already keeps it out of the asset hash (so its
removal is asset-hash-neutral).
5. If Phase 4 has landed (bundling moved into a cdk/common helper) note the
new location.
12-20 facts.`,
{ label: 'recon:cdk-bundling', model: 'haiku', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only AWS + docs recon for the runbook (region us-east-1, READ-ONLY):
1. Both email-processor DLQs: exact SQS queue names/ARNs and the
MessageRetentionPeriod (confirm the ~14-day breadcrumb window). Find them
via the stacks (dead_letter_queue / make_processor_dlq) and
aws sqs get-queue-attributes. Also the async event-invoke config /
on-failure destination if present (get-function-event-invoke-config) —
confirm it is an async on-failure destination DLQ with NO console
redrive-to-source.
2. Both raw-email S3 buckets: exact names and the inbound/ prefix lifecycle
rule (get-bucket-lifecycle-configuration) — confirm the ~90-day expiry
that overrides the table RETAIN policy (S3 is the replay floor).
3. The relevant CloudWatch alarm names the runbook references: the
sender-auth-rejected alarm(s) and the ai_fallback_rejected alarm(s) per
pipeline (describe-alarms / grep the stacks) — so the runbook states that
those two drop classes are fail-closed SKIPS that never reach the DLQ.
4. Quote the README alarms section (its heading + surrounding lines) so the
Implement agent can add the runbook link in the right place, matching
style.
Return names/ARNs/windows as facts; flag any name you could not resolve so
the runbook can be validated against real infra later.`,
{ label: 'recon:dlq-alarms', model: 'sonnet', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n')
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
.filter(b => b && !/^\s*(none|n\/a)\b/i.test(b))
log(`Recon complete: ${reconOk.length}/4 mappers, ${reconBlockers.length} blockers`)
if (reconBlockers.length) {
return { aborted: 'recon blockers (likely a surviving third-party dep blocking cp-only, or an unresolvable pipeline name)', blockers: reconBlockers, reconPack: pack }
}
// -------------------------------------------------------------------- spec
phase('Spec')
const spec = await agent(`${PREAMBLE}
You are the SPEC agent — the single authority that pins every contested
decision BEFORE parallel implementation (parallel leaves cannot see each
other's choices). Using the recon pack below plus your own reads of the
actual files, produce the binding implementation spec:
- reprocessCli: the complete generalized scripts/reprocess.py — argparse
surface (--pipeline po|wo required; --key / --prefix / --since targeted
modes; --all for full-prefix; --execute preserving dry-run-by-default),
the per-pipeline function-name + bucket resolution table, and the EXACT
--help + in-code caveat text for --all (all five caveats from constraint 1
verbatim). TARGETED is the default path; --all is the ONLY way to sweep the
whole prefix.
- contractTest: the synthetic-event-shape test — file path (collected by the
standard pytest run), the assertions on Records[0].s3.bucket.name and
object.key, and the raw-key / no-URL-decoding assertion with a concrete
encoded-vs-raw example (e.g. a key with a space or '+').
- runbook: the full docs/runbook-dlq-recovery.md outline with the real
DLQ/queue/alarm/bucket names from recon, the no-redrive procedure steps
(receive -> extract S3 key from event body -> targeted re-invoke via
reprocess.py --key -> verify -> purge), the 14d-DLQ / 90d-S3 windows +
lifecycle-overrides-RETAIN note, the intentional sender-auth +
ai_fallback_rejected non-DLQ statement, and the exact README alarms-section
link edit.
- dependencyEdits: exact new contents of every manifest — both
email-processor requirements.txt (handbook empty-with-comment), wo/web_ui
(empty-with-comment), po/web_ui + po/site_extractor (NEW; decide pinned dep
vs whether the dependabot entry is worthwhile — a dependabot entry needs at
least one pinned dep to act on, so pin boto3 exactly if you want the entry
live), tests/ (moto==<resolved version> from recon) — plus the exact
dependabot.yml additions (three entries) matching the existing shape.
- bundlingEdits: both email-processor bundling command strings BEFORE and
AFTER, verbatim — the cp-only result, an explicit proof line that no
third-party dep remains (so removing the whole pip step is safe, NOT a
regression of the manylinux pin), and confirmation the exclude lists +
widened '../lambdas' root + any Phase-3 'cp shared/*.py' line survive
untouched. If Phase 4 moved bundling into a helper, target that location.
- packageCleanup: the exact rm -rf command for
lambdas/po/email_processor/package/, the note that it is a filesystem rm
(UNTRACKED — never git rm) performed and REPORTED by the Package agent, not
committed, and why it is asset-hash-neutral (Phase 2 exclude).
- verifyRules: exactly how Verify judges success — the pytest set incl. the
new contract test, ruff now INCLUDING scripts/, both cdk synth,
dependabot.yml yaml-parse, requirements resolve/import, and the runtime
'import boto3' proof for a cp-only staged asset.
Recon pack:\n${pack}`,
{ label: 'spec:pin-ops-recovery', phase: 'Spec', schema: SPEC })
if (!spec) return { aborted: 'spec agent died — rerun workflow', reconBlockers }
const specBlock = `BINDING SPEC (from the spec agent — implement EXACTLY this):\n${JSON.stringify(spec, null, 2)}`
log('Spec pinned: reprocess CLI + caveats, contract test, runbook, dependency edits, cp-only bundling, package cleanup')
// --------------------------------------------------------------- implement
phase('Implement')
const impl = await parallel([
() => agent(`${PREAMBLE}
YOU OWN: scripts/reprocess.py and the new synthetic-event contract test file
ONLY (per spec.contractTest's path — do not touch any other test). Do not
touch cdk/, requirements, dependabot, docs, or README.
Task: rewrite scripts/reprocess.py per spec.reprocessCli — add --pipeline
po|wo, --key/--prefix/--since (TARGETED, the default), --all (DEMOTED
full-prefix, with all five caveats in BOTH --help and code comments),
preserving dry-run-by-default / --execute. Keep build_s3_event emitting the
RAW key (no URL-decoding). Then add the contract test per spec.contractTest
pinning the event shape + the raw-key / no-double-decode assertion.
Run before returning: ruff check scripts && ruff format scripts --check,
python3 scripts/reprocess.py --help (must render the caveats), and
pytest -q --no-cov -k <the contract test> (tests land in parallel — if
collection of other roots is red from another agent's in-flight edit, run
your test file directly; poll up to ~10 min before reporting a blocker).
${specBlock}`,
{ label: 'impl:reprocess', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: every requirements.txt (lambdas/po/email_processor,
lambdas/wo/email_processor, lambdas/wo/web_ui, tests/, and the two NEW
manifests lambdas/po/web_ui/requirements.txt +
lambdas/po/site_extractor/requirements.txt), .github/dependabot.yml, and the
bundling regions of cdk/po_stack.py + cdk/wo_stack.py ONLY (only the
email-processor from_asset bundling command — alarms, IAM, tables, env,
memory, timeout, the three plain from_asset calls are all off-limits).
Task: apply spec.dependencyEdits (drop vendored boto3 -> empty-with-comment
in both email-processor + wo/web_ui; create the two pinned po manifests;
exact-pin moto==) + the three dependabot entries + spec.bundlingEdits
(cp-only: remove ONLY the pip step from both email-processor commands,
preserving the exclude lists, widened root, and any Phase-3 shared cp line).
Do NOT delete package/ (that is the Package agent's reported step).
Run before returning: ruff check cdk, python3 -c 'import yaml,pathlib;
yaml.safe_load(pathlib.Path(".github/dependabot.yml").read_text())' (parses),
cd cdk && npx cdk synth po-ingest -q -o /tmp/phase7-synth &&
npx cdk synth workorder-ingest -q -o /tmp/phase7-synth (artifact-id
selectors). Docker bundling runs — confirm each staged email-processor asset
still contains handler.py + its siblings (and shared/*.py if Phase 3 landed)
and that 'import boto3' resolves at runtime from the Lambda runtime copy (the
staged asset need not vendor it). Note asset hashes. If the other agents'
edits have not landed the synth is unaffected (bundling is yours) — but poll
~10 min if Docker is slow before a blocker.
${specBlock}`,
{ label: 'impl:deps-bundling', model: 'sonnet', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: docs/runbook-dlq-recovery.md (NEW) and README.md ONLY.
Task A: write docs/runbook-dlq-recovery.md per spec.runbook — the no-redrive
async-destination DLQ procedure (receive-message -> extract the S3 key from
the event body -> targeted re-invoke via 'python scripts/reprocess.py
--pipeline <po|wo> --key <key> --execute' -> verify the write -> purge the
message), the real queue/bucket/alarm names from the spec, the recovery
windows (14-day DLQ breadcrumb, 90-day raw-email S3 with the
lifecycle-overrides-RETAIN note), and the explicit statement that sender-auth
rejections AND ai_fallback_rejected drops are fail-closed SKIPS that never
reach the DLQ (no retry, no DLQ message).
Task B: README — link the new runbook from the alarms section per
spec.runbook's exact edit, and note reprocess.py is now pipeline-general
(--pipeline / --key / --prefix / --since targeted; --all demoted). Match
existing README style.
Run before returning: confirm the runbook links resolve and the README edit
is in the alarms section. (No pytest owned by you; if you want a doc-lint,
ruff does not lint .md.)
${specBlock}`,
{ label: 'impl:runbook-readme', model: 'opus', phase: 'Implement', schema: IMPL }),
])
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/3 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'scripts/reprocess.py',
'tests/',
'docs/runbook-dlq-recovery.md',
'lambdas/po/email_processor/requirements.txt',
'lambdas/wo/email_processor/requirements.txt',
'lambdas/wo/web_ui/requirements.txt',
'lambdas/po/web_ui/requirements.txt',
'lambdas/po/site_extractor/requirements.txt',
'.github/dependabot.yml',
'cdk/po_stack.py',
'cdk/wo_stack.py',
'README.md',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification — trust nothing self-reported. Run ALL gates,
quoting failures verbatim:
1. pytest -q --no-cov (repo root, all roots green) — INCLUDING the new
synthetic-event contract test (name it and quote its pass line).
2. ruff check . && ruff format --check . — scripts/ is now IN scope (it was
previously omitted); confirm scripts/reprocess.py lints and formats clean.
3. cd cdk && npx cdk synth po-ingest -q && npx cdk synth workorder-ingest -q
(Docker bundling runs; both must succeed).
4. cd cdk && npx cdk diff po-ingest ; npx cdk diff workorder-ingest —
constraint 9: the ONLY resource delta allowed is the two email
processors' Code/S3Key (asset) + CDK metadata. Any alarm/IAM/env/runtime/
memory/timeout/handler-prop/logical-ID delta = FAIL. Paste the diff
summaries.
5. DEPENDABOT: python3 -c 'import yaml,pathlib; d=yaml.safe_load(pathlib.Path(".github/dependabot.yml").read_text()); print([u["directory"] for u in d["updates"]])'
— the three new dirs (/tests, /lambdas/po/web_ui, /lambdas/po/site_extractor)
present, shape matches existing entries.
6. MANIFESTS: every email-processor + wo/web_ui requirements.txt is the
empty-with-comment form (no pinned third-party dep); tests/requirements.txt
is moto==<exact>; po/web_ui + po/site_extractor manifests EXIST and are
PINNED (a dependabot entry pointed at a floor/absent manifest is a no-op —
FAIL if unpinned). pip install --dry-run (or pip-compile check) resolves
each without error.
7. RUNTIME IMPORT: no lambda handler .py under lambdas/*/email_processor
changed (git diff ${BASE}...HEAD -- must be empty for those); 'import boto3'
still resolves — prove the Lambda runtime copy suffices (the cp-only staged
asset need not vendor boto3).
8. RUNBOOK: docs/runbook-dlq-recovery.md exists, states the 14d/90d windows,
the no-redrive procedure, and the sender-auth + ai_fallback_rejected
non-DLQ note; README alarms section links it.
9. git status --porcelain scope check: every modified/added path under
${EXPECTED_SCOPE.join(', ')} (untracked .coverage/.claude/package/
tolerated; package/ must NOT be staged).
passed=true only if all green. YOU MAY NOT edit files.`
const lenses = [
{ key: 'bundling-integrity', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — bundling-integrity lens. The bundling was simplified to
cp-only and vendored boto3 was dropped; try to prove a runtime ImportError
was introduced (the PR #34 / PR #105 failure class). (1) cd cdk && synth both
stacks; inside each staged email-processor asset dir list the files and run
python3 -c "import handler" with that dir alone on sys.path (env-var stubs as
needed) — does every runtime-imported module still ship (handler siblings +
shared/*.py if Phase 3 landed)? (2) Does 'import boto3' succeed against the
Lambda runtime copy — i.e. is boto3 genuinely NOT needed in the zip, or does
some code path need a newer boto3 than the runtime ships? (3) Are the Phase
2/3 excludes (['**/__pycache__/**','**/tests/**','**/package/**']) STILL
present and is the widened '../lambdas' root intact? (4) Was the manylinux
pin removed ONLY as part of removing the whole pip step (because nothing is
installed), NOT while keeping a pip install? A cp-only command that still
tries to pip-install without the pin, or a surviving third-party dep, is a
critical finding. (5) DETERMINISM: synth po-ingest twice into fresh -o dirs —
identical asset hashes. confirmed=true only with a concrete ImportError
sketch or file:line proof.` },
{ key: 'reprocess-safety', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — reprocess-safety lens. Attack the generalized
reprocess.py. (1) Is the DEFAULT behavior TARGETED (—key/—prefix/—since), and
is it IMPOSSIBLE to sweep the whole inbound/ prefix without the explicit
--all flag? Trace the arg parsing — can a missing/empty --prefix silently
fall through to a full-prefix replay? That would be a critical finding
(accidental mass re-invoke = mass Bedrock re-bill + merged-field regression).
(2) Are ALL FIVE --all caveats present in BOTH --help and code comments
(concurrency/no-serialize, RequestResponse-if-order-matters, double-count,
Bedrock re-bill, out-of-order regression)? (3) Does build_s3_event emit the
RAW key with NO URL-decoding, and does the contract test actually FAIL if
someone adds an unquote_plus (mutate it on a scratch copy and confirm red)?
A key with a space/'+' must round-trip raw so the handler decodes exactly
once. (4) --pipeline wo resolves the correct function name + bucket (not the
PO defaults)? confirmed=true only with file:line or reproduced-failure
evidence.` },
{ key: 'dependency', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — dependency lens. (1) moto: is tests/requirements.txt
EXACT-pinned (moto==<version>), not a floor (>=)? A floor makes the tests/
dependabot entry a no-op. (2) The three NEW dependabot dirs: do
/lambdas/po/web_ui and /lambdas/po/site_extractor now have PINNED manifests
that actually exist? A dependabot entry pointed at an absent or floor-pinned
manifest does nothing — that is a finding. Cross-check dependabot.yml
directory strings against the real file paths (a typo'd directory silently
no-ops). (3) The email-processor + wo/web_ui requirements are the handbook
empty-with-comment form — confirm no pinned third-party dep is left that
would then require the pip step back (which would contradict the cp-only
bundling). (4) Does the empty-with-comment wording match lambda-template.md
(runtime provides boto3)? confirmed=true only with file:line evidence.` },
]
let round = 0
let checks = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
confirmed = results.slice(1).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && f.severity !== 'low')
const green = checks && checks.passed
log(`Verify round ${round}: mechanical ${checks && checks.passed ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (green && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit files under: ${EXPECTED_SCOPE.join(', ')}.
Fix EVERY item below minimally; the binding spec and 10 pinned constraints
still hold (a finding that conflicts with a constraint is reported, not
"fixed" — the constraint wins, esp. constraint 5's cp-only-only-if-no-dep
rule, constraint 8's do-NOT-git-rm-package, and constraint 9's no-handler /
no-alarm change). Re-run the specific failing gate/test per fix.
MECHANICAL:\n${checks ? checks.details : '(agent died — rerun all gates)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}
${specBlock}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
spec,
}
}
// ----------------------------------------------------------------- package
phase('Package')
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent:
1. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md
and follow it exactly.
2. OPTIONAL LOCAL CLEANUP (constraint 8): rm -rf
lambdas/po/email_processor/package/ — it is UNTRACKED, so this is a
filesystem rm, NOT git rm; there is NOTHING to stage from it and it is
asset-hash-neutral (Phase 2 exclude). PERFORM it and REPORT the reclaimed
~44 MB in your summary. It is NOT part of the commit.
3. git add only paths under: ${EXPECTED_SCOPE.join(', ')} and
.claude/workflows/phase-7-ops-recovery.js. NOT .coverage, NOT package/
(it is gone / untracked either way). Verify the staged set with
git status — the two NEW manifests and the new runbook MUST be staged.
4. ONE commit; write the message to /tmp/phase7-commit-msg.txt and use
git commit -F /tmp/phase7-commit-msg.txt (backticks in -m get eaten by
zsh). Suggested subject:
"feat: ops/recovery tooling + dependency hygiene — generalized reprocess, DLQ runbook, cp-only bundling (refactor phase 7)"
Body: the reprocess generalization (targeted-primary, --all demoted with
caveats) + contract test, the DLQ runbook + windows, the dropped vendored
boto3 / cp-only bundling, the moto exact-pin + three dependabot entries
with pinned manifests, and the reported (not committed) package/ cleanup.
NO AI attribution / Co-Authored-By lines.
5. Do NOT push. Return commit sha + shortstat + the package/ cleanup result
in summary.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
base: BASE,
commit: commit ? commit.summary : 'commit agent died — commit manually',
spec: { reprocessCli: spec.reprocessCli, runbook: spec.runbook, dependencyEdits: spec.dependencyEdits, bundlingEdits: spec.bundlingEdits, packageCleanup: spec.packageCleanup },
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
outstandingGates: [
'/sh-security-review NOT required (ops tooling + dependency hygiene only — no untrusted-input parsing, no auth code, no IAM/policy change). The pre-push deterministic scanners still run as the unattended backstop.',
'cross-family cross_review.py NOT required (no IAM/policy change; no Lambda handler signature / event-shape / return-contract change — reprocess.py emits the SAME S3 event shape it always did, pinned by the new contract test).',
'push + PR + gh pr checks green',
'deploy-then-merge for the bundling/dep changes: deploy from branch, smoke green, one real email per pipeline (the dropped-vendored-boto3 email processors MUST still import boto3 at runtime from the Lambda runtime copy), expect ONE benign asset-hash redeploy from the package/ cleanup + cp-only change, THEN merge.',
'reprocess.py + docs/runbook-dlq-recovery.md are ops/docs (no deploy risk), but the runbook should be VALIDATED against the real DLQ / alarm / bucket names before it is relied on in an incident.',
'Confluence "AWS Architecture Map" — Phase 7 adds the DLQ-recovery runbook to the inventory; note the new runbook when the resource inventory is next refreshed.',
],
}

View file

@ -1,712 +0,0 @@
export const meta = {
name: 'phase-8-test-consolidation',
description: 'Phase 8 of the procurement-ingest refactor (docs/refactor-evaluation.md §4): test-root consolidation — keep BOTH roots but fix the LOADING via ONE new repo-root conftest.py (dummy AWS env + moto BUILTIN_HANDLERS before any handler import + single load_lambda_module keeping one sys.modules save/restore); a shared tests/support/ package (superset FakeTable, FakeDynamoResource, load_email, load_golden with parse_float=Decimal); rewrite _wo_parser_support.py off the bare import strategy; move test_po_merge/test_pad_zip into the PO tests dir (test_parse_raw_email/test_ses_auth stay root); delete test_local.py; add the missing scenarios (PO+WO Bedrock transport errors, handler SES-auth reject seam, web_ui both, WO merge semantics, small pins); CI gains --cov with an explicit module list (or __init__.py), a fail-under, the Phase 0 AST bundle test, an enforced ruff C901/PLR config incl scripts/. The ONLY product-code change is the WO handler Bedrock-error metric-wrap (except-and-reraise, no double-count) + invalid_status reason-code fix, and the PO _validate_new_po_values per-rule split. LAST PR — validates the new module boundaries, so it hard-gates on Phases 3 AND 5 merged. Committed locally, never pushed (push gated on /sh-security-review in the main loop for the WO metric change).',
phases: [
{ title: 'Setup', detail: 'verify Phases 3 (lambdas/shared/) AND 5 (decomposed handler siblings) on base, branch feature/phase-8-test-consolidation', model: 'haiku' },
{ title: 'Recon', detail: '4 mappers: the three loaders + FakeTable/support duplication, test inventory + coverage gaps, WO metric/reason-code + PO validation-split surface, CI/lint/ruff/README drift' },
{ title: 'Spec', detail: 'serial fable spec: pin root conftest, support package, loader rewrites, file moves/deletions, new scenarios, WO metric-wrap + reason-code, PO validation split, ruff config, CI changes, README, disjoint ownership' },
{ title: 'Implement', detail: 'opus: all tests + support + root conftest; opus: WO handler + PO template_parser product-code + newly-surfaced lint fixes; sonnet: ruff/pytest/CI config + README — disjoint files', model: 'opus' },
{ title: 'Verify', detail: 'mechanical gates (full suite green under the new loader, goldens UNCHANGED, ruff green incl scripts/, both cdk synth, --cov covers web_ui/site_extractor, WO no-double-count pin passes) + 3 fable lenses (loader-integrity, WO-metric-contract, coverage-honesty)' },
{ title: 'Fix', detail: 'opus fixer, full re-verify, max 3 rounds', model: 'opus' },
{ title: 'Package', detail: 'single commit via -F (no push)', model: 'sonnet' },
],
}
// ---------------------------------------------------------------- constants
const REPO = '/Users/adammoussa/Documents/repositories/seahaven/procurement-ingest'
const BRANCH = 'feature/phase-8-test-consolidation'
let _args = args
if (typeof _args === 'string') {
try { _args = JSON.parse(_args) } catch (e) { _args = null }
}
const BASE = (_args && _args.base) || 'main'
const CONSTRAINTS = `
PINNED BEHAVIORAL CONSTRAINTS (docs/refactor-evaluation.md Phase 8 + §4 — violating any is a build failure):
1. KEEP BOTH TEST ROOTS; fix the LOADING, not the split. ONE loader: a NEW
repo-root conftest.py (tests/conftest.py does NOT load for a standalone
'pytest lambdas/po' run, so it cannot carry session invariants) that:
sets the dummy AWS env; registers moto BUILTIN_HANDLERS BEFORE any handler
import (carry the explanatory comment currently buried at
_po_parser_support.py:33 verbatim — boto3 sessions only pick up moto's
stubber hook if created AFTER registration); and exposes a single
load_lambda_module(pipeline, name) keeping ONE copy of the sys.modules
save/restore dance (tests/conftest.py:70-85). That dance does NOT shrink to
nothing — template_parser is still a duplicated bare name across pipelines
and still needs per-exec sibling binding.
2. Rewrite _wo_parser_support.py OFF the bare 'import handler' /
'from handler import parse_raw_email' strategy (it is the source of the
bare-name sys.modules collision the other two loaders defend against — see
_wo_parser_support.py:17-18,39). Add a shared tests/support/ package with:
a SUPERSET FakeTable (PO's update_item recording + WO's put_item and the
keyed single-row store from _wo_parser_support.py:51-67), FakeDynamoResource,
load_email, load_golden — KEEPING PO's parse_float=Decimal in load_golden
(load-bearing: exact money comparison at PO magnitudes;
_po_parser_support.py load_golden). PO-only helpers keep Decimal; WO's
load_golden has no parse_float today — the superset must not regress PO.
3. FILE MOVES: tests/test_po_merge.py and tests/test_pad_zip.py ->
lambdas/po/email_processor/tests/ (PO-specific, belong beside the code).
tests/test_parse_raw_email.py and tests/test_ses_auth.py STAY at root
(genuinely cross-pipeline, parameterized over BOTH handlers). git mv so
history follows; fix their imports to the new support package.
4. DELETE tests/../test_local.py (repo root: globs a nonexistent samples/,
WO-only, imports a handler at collection so it bypasses the loader gate;
the golden suites cover its role). Deletion is PREFERRED over a --pipeline
rewrite. Its pytest.ini exclusion comment goes with it.
5. ADD the missing scenarios by module (§4 priority order), using the new
single loader + support package for EVERY new test:
(a) PO+WO Bedrock transport errors: ThrottlingException, missing 'content'
key, empty content list, non-JSON model text. Assert PO's pre-call
ai_fallback metric survived + no partial write + exception propagates.
PIN WO's no-datapoint-on-throttle behavior with a DOCUMENTING test —
do NOT "fix" it by reordering (constraint 7 owns the real fix).
(b) Handler-level SES-auth reject seam, per pipeline: NO auth monkeypatch +
empty ALLOWED_DKIM_DOMAINS -> assert ZERO Bedrock calls, ZERO writes,
NO raise (env read at call time; FakeS3 still needed, gate sits after
get_object). Today deleting the gate line passes all tests — this closes
that hole.
(c) web_ui BOTH functions (0% today; auth is the mandatory-review surface):
fail-closed on unset ARN (module reload — ARN read at import),
fail-closed on a Secrets-Manager exception, TTL cache refresh,
Bearer / X-Auth-Token / header case-insensitivity, wrong-token 401,
401 WITHOUT a table scan, non-ASCII token, hostile-field escaping
regression lock. PO web_ui lacks __init__.py — use the loader, not
package imports.
(d) WO merge semantics: a moto-backed mirror of test_po_merge (table name
'WorkOrders', NOT kebab): null-status never clobbers wo_status,
created_at immutable via if_not_exists, status->wo_status mapping, None
fields absent from SET, record_type only-when-present.
(e) Small pins: PO-DC-02 64-char EMF clamp regression, per-pipeline
multi-record failure-isolation (all-or-retry contract), reprocess.py
synthetic-event-shape contract IF Phase 7 has not already added it
(recon confirms — do not duplicate).
6. CI CHANGES (in the same PR): add --cov with an EXPLICIT module list (OR add
__init__.py so --cov=lambdas stops silently skipping web_ui/site_extractor
for the missing package marker) — pick ONE, state why; add a fail-under
ONCE the web_ui/site_extractor suites exist; keep the Phase 0 AST
bundle-consistency test in the standard pytest run; ADD a ruff config
(pyproject.toml or ruff.toml) that ENABLES C901/PLR so the complexity
ceilings are ENFORCED not decorative; include scripts/ in the lint scope;
keep the strict 'ci / ci' required check; NO admin-bypass pushes.
7. WO HANDLER PRODUCT-CODE CHANGES (the ONLY product-code deltas besides the PO
split): (a) invalid_status reason-code fix — GREP the dashboards/metric
filters for 'malformed_site_code' FIRST and report before renaming (one
status failure currently yields two codes by path). (b) WO Bedrock-error
metric fix: WRAP the Bedrock call so ai_fallback + bedrock_error are emitted
in an except-and-RERAISE. This is NOT a naive reorder — a reorder emits the
metric unconditionally pre-call and DOUBLE-COUNTS gate-rejected emails
against wo_stack's "a rejected email emits nothing else" alarm contract.
The handler EVENT/RETURN CONTRACT is unchanged (no signature change). PIN
the no-double-count with a test that computes the emitted series by hand.
8. PO _validate_new_po_values PER-RULE SPLIT (template_parser.py): break the
monolith into per-rule helpers; the V4 anchor-frame dataclass must CARRY
summary_matches / price so V13 can consume them; include extract_new_po
(C901=35) in the split scope. The NEW ruff PLR/C901 config makes the
previously-INERT 'noqa: PLR09xx' suppressions LIVE — every newly-surfaced
violation across the tree must be FIXED or noqa'd-with-written-justification
IN THIS SAME PR. EXCEPTION: files under the shadow-bake freeze
(derived_fields.py) are UNTOUCHABLE — surface their violations via a
per-file ignore in the ruff config (with a justification comment), NEVER an
in-file edit. Docstring/README drift batch (findings 33-38) lands here too.
9. UNTOUCHABLE (git diff ${BASE}...HEAD must be empty for each): every product
module EXCEPT lambdas/wo/email_processor/handler.py (constraint 7) and
lambdas/po/email_processor/template_parser.py (constraint 8). Specifically
derived_fields.py (shadow bake), both ses_auth, both validate_ai_fallback
gates, lambdas/shared/*, all CDK stacks, site_extractor. GOLDEN fixtures
(tests/**/expected/*.json and the .eml corpora) are byte-frozen — a test
that only passes after a golden edit is a FAIL. New tests must NOT edit
existing goldens; WO's one SES-stamped fixture (synthesized header block)
is the only new fixture allowed.
10. This is the LAST phase: the resulting suite / coverage fail-under / ruff
config become the NEW CI floor. Do not weaken any existing gate to make a
new one pass.
`
const PREAMBLE = `
You are one of several agents building refactor Phase 8 in the git repo at
${REPO} on branch ${BRANCH} (already checked out — do NOT switch branches,
do NOT create branches, do NOT commit, NEVER push, do NOT run cdk deploy or
touch AWS resources beyond read-only calls).
Authoritative spec: docs/refactor-evaluation.md, section "Phase 8 —
Test-root consolidation" and the whole of "§4 Test-hardening plan". The doc
wins on any conflict.
Work ONLY in the files you are told you own; other agents are concurrently
editing other files in this same working tree.
${CONSTRAINTS}
Your final message is consumed by an orchestrator script, not a human —
return only the structured data requested.
`
// ------------------------------------------------------------------ schemas
const RECON = {
type: 'object',
required: ['summary', 'facts'],
properties: {
summary: { type: 'string' },
facts: { type: 'array', items: { type: 'string' } },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const SPEC = {
type: 'object',
required: ['rootConftest', 'supportPackage', 'loaderRewrites', 'fileMoves', 'newScenarios', 'productCodeEdits', 'ruffConfig', 'ciChanges', 'readmeDocs', 'ownership', 'notes'],
properties: {
rootConftest: { type: 'string', description: 'the complete new repo-root conftest.py design: dummy AWS env block, the moto BUILTIN_HANDLERS-before-handler-import registration with the carried :33 comment, and the single load_lambda_module(pipeline, name) signature keeping ONE sys.modules save/restore (state exactly why it does not shrink to nothing — template_parser bare name). How tests/conftest.py, _po_parser_support.py and _wo_parser_support.py reconcile against it (deleted / shimmed / re-pointed).' },
supportPackage: { type: 'string', description: 'tests/support/ package contents: the SUPERSET FakeTable field-by-field (PO update_item record + WO puts/keyed store), FakeDynamoResource, load_email, load_golden — explicitly pinning parse_float=Decimal is kept and that WO callers gain it without regressing (are WO goldens integer-only?). __init__.py exports.' },
loaderRewrites: { type: 'string', description: 'exact per-file edits to retire the bare-import strategy in _wo_parser_support.py and rewire both support files + tests/conftest.py onto load_lambda_module; the moto-before-handler ordering must survive; which fixtures (email_handler, ses_auth, po_handler, wo_handler) move where.' },
fileMoves: { type: 'string', description: 'the git mv list (test_po_merge.py, test_pad_zip.py -> PO tests dir), the stay-at-root set (test_parse_raw_email.py, test_ses_auth.py), the test_local.py deletion + its pytest.ini comment removal, and the import rewrites each moved file needs.' },
newScenarios: { type: 'string', description: 'per-module new test files (paths) and the scenarios each covers per constraint 5 (a-e): Bedrock transport errors both pipelines, SES-auth reject seam both pipelines, web_ui both functions, WO merge semantics, the small pins — and whether the reprocess synthetic-event pin already exists from Phase 7.' },
productCodeEdits: { type: 'string', description: 'the WO handler edit (invalid_status reason-code fix + Bedrock-error except-and-reraise metric-wrap with the exact emitted series proving NO double-count vs wo_stack alarm contract; the malformed_site_code dashboard-grep result) and the PO _validate_new_po_values / extract_new_po per-rule split (V4 dataclass carries summary_matches/price for V13). file:line for each.' },
ruffConfig: { type: 'string', description: 'the exact new ruff config (pyproject.toml or ruff.toml): which rule families (C901/PLR), the complexity ceilings (extract_new_po C901=35), lint scope incl scripts/, and the per-file-ignore for derived_fields.py with justification. The full list of newly-surfaced violations and their fix-or-noqa disposition.' },
ciChanges: { type: 'string', description: 'the exact .github/workflows/ci.yaml edits: --cov approach (explicit module list vs __init__.py, with the reason), where the fail-under lands and its value, keeping the AST bundle test + strict ci/ci check, scripts/ lint scope, no admin bypass.' },
readmeDocs: { type: 'string', description: 'README + docstring drift batch (findings 33-38): exact sections/lines to fix, matched to existing style.' },
ownership: { type: 'string', description: 'the DISJOINT file-ownership map for the 3 parallel implement agents (tests+support+root-conftest / product-code+newly-surfaced-lint / config+CI+README) — no path owned by two agents; note the ruff-config -> product-lint dependency and how the poll resolves it.' },
notes: { type: 'string' },
},
}
const IMPL = {
type: 'object',
required: ['filesChanged', 'summary', 'checksRun'],
properties: {
filesChanged: { type: 'array', items: { type: 'string' } },
summary: { type: 'string' },
checksRun: { type: 'string' },
blockers: { type: 'array', items: { type: 'string' } },
},
}
const CHECKS = {
type: 'object',
required: ['passed', 'details'],
properties: {
passed: { type: 'boolean' },
details: { type: 'string' },
scopeViolations: { type: 'array', items: { type: 'string' } },
},
}
const FINDINGS = {
type: 'object',
required: ['findings'],
properties: {
findings: {
type: 'array',
items: {
type: 'object',
required: ['title', 'severity', 'confirmed', 'evidence', 'fix'],
properties: {
title: { type: 'string' },
severity: { enum: ['critical', 'high', 'medium', 'low'] },
confirmed: { type: 'boolean' },
evidence: { type: 'string' },
fix: { type: 'string' },
},
},
},
},
}
// ------------------------------------------------------------------- setup
phase('Setup')
const setup = await agent(`
In ${REPO}:
1. SEQUENCING HARD-GATE — this is the LAST PR and it validates the NEW module
boundaries, so Phases 3 AND 5 must both be on ${BASE} (shared/ from Phase 3
is what the loader now resolves against; the decomposed handler siblings
from Phase 5 are the boundaries these suites exercise). git fetch origin,
then pick the base ref: origin/${BASE} if that remote ref exists, otherwise
the local branch ${BASE} (a stacked local-only base is expected and fine).
Verify on the base ref:
(a) Phase 3: lambdas/shared/ exists with its modules
(git ls-tree <baseref> -- lambdas/shared/ — must list ses_auth.py etc.);
(b) Phase 5: the decomposed PO handler siblings exist
(git ls-tree <baseref> -- lambdas/po/email_processor/ must include
extraction.py, enrichment.py, telemetry.py, persistence.py, prompts.py).
If EITHER is missing, STOP with a blocker naming the unmet phase and do
nothing else.
2. Verify clean working tree (untracked .coverage / .claude/ / the local
44 MB lambdas/po/email_processor/package/ dir are fine; any OTHER dirt =
blocker, never stash or discard).
3. git checkout ${BASE}; then git pull --ff-only ONLY if the branch has an
upstream (a local-only base skips the pull — not a blocker); then
git checkout -b ${BRANCH}
4. gh pr list --state open --json number,title,headRefName (overlap check).
Return facts: HEAD sha, per-phase gate evidence (the two git ls-tree outputs),
open PRs, blockers.
`, { label: 'setup:branch', model: 'haiku', schema: RECON })
if (!setup || (setup.blockers && setup.blockers.length)) {
return { aborted: 'setup blockers', blockers: setup ? setup.blockers : ['setup agent died'], facts: setup ? setup.facts : [] }
}
log(`Branch ${BRANCH} ready off ${BASE}. ${setup.summary}`)
// ------------------------------------------------------------------- recon
phase('Recon')
const recon = await parallel([
() => agent(`${PREAMBLE}
Read-only recon of the THREE loading idioms + the fake-Dynamo helpers this
phase collapses:
1. tests/conftest.py — quote the importlib loader, the _SIBLING_MODULES tuple
(ses_auth/template_parser/derived_fields), and the sys.modules save/restore
dance (lines ~70-85). Note that after Phase 5 the PO sibling set changed
(extraction/enrichment/telemetry/persistence/prompts) — quote the CURRENT
sibling list on this branch's base, it may already differ from the audit.
2. _po_parser_support.py — the independent importlib reimplementation, the
load-bearing moto-before-handler comment at line 33 (quote it verbatim, it
must be carried into the new root conftest), load_golden's parse_float=Decimal
(quote it), and its FakeTable (update_item only).
3. _wo_parser_support.py — the bare sys.path 'import handler' /
'from handler import parse_raw_email' strategy (lines 17-18, 39), and its
FakeTable (put_item + keyed store, lines 51-67) — the superset target.
4. pytest.ini — the testpaths (three roots) and the test_local.py exclusion
comment. Confirm test_local.py exists at repo root and what it imports at
collection.
Diff the two FakeTable classes field-by-field to define the superset. Confirm
whether WO's load_golden uses parse_float (it does not today) and whether any
WO golden is non-integer (would break under Decimal — pin it).
20-30 precise facts.`,
{ label: 'recon:loaders-support', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the test INVENTORY + coverage gaps this phase must fill:
- Enumerate every test_*.py across all three roots (tests/,
lambdas/po/email_processor/tests/, lambdas/wo/email_processor/tests/) with a
one-line purpose each. Flag test_po_merge.py + test_pad_zip.py at root (move
targets) and test_parse_raw_email.py + test_ses_auth.py at root (stay).
- For the MISSING scenarios (§4), confirm the current gap and the seam each new
test hooks: (a) PO+WO Bedrock transport-error handling — quote where each
handler invokes Bedrock and where ai_fallback is emitted relative to the call
(PO pre-call; WO post-gate); (b) the SES-auth gate call site in each handler
(authenticate_inbound_email) and how existing dispatch tests monkeypatch it;
(c) web_ui — BOTH functions' auth entry points, whether they have any tests
or __init__.py today (they do not), how the ARN/Secrets-Manager fail-closed
path reads config (import-time vs call-time), the table-scan path a 401 must
avoid; (d) WO save_work_order merge semantics (table 'WorkOrders'); (e) the
small pins — the 64-char EMF clamp site, multi-record event loop, and whether
scripts/reprocess.py already has a synthetic-event contract test from Phase 7
(git log / grep — do NOT duplicate it).
- Confirm the golden corpora locations (expected/*.json, .eml dirs) that are
byte-frozen.
25-35 facts.`,
{ label: 'recon:inventory-gaps', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the TWO product-code change surfaces (constraints 7-8):
1. WO handler (lambdas/wo/email_processor/handler.py): quote with file:line the
Bedrock invoke, the emit_parse_metric / ai_fallback emission, and the gate
that returns 'invalid_status' vs 'malformed_site_code'. Establish the CURRENT
emitted-metric series for a gate-rejected email and for a Bedrock-error email
— enough to prove the except-and-reraise wrap does NOT double-count. THEN
grep the whole repo (cdk/*.py, dashboards, metric filters, docs) for
'malformed_site_code' and 'invalid_status' and list every consumer — the
reason-code rename must not silently break a dashboard/alarm. Quote the
wo_stack "a rejected email emits nothing else" alarm/MathExpression it must
not violate.
2. PO template_parser.py: quote _validate_new_po_values (the ~294-line monolith,
its 'noqa: PLR09xx' suppressions) and extract_new_po (C901=35). Identify the
V4 anchor-frame dataclass and what fields it carries today vs the
summary_matches/price V13 needs. Confirm both currently pass lint ONLY
because no ruff config selects PLR/C901.
Return the exact emitted-series tables, the malformed_site_code consumer list,
and the validation-split surface. 20-30 facts.`,
{ label: 'recon:product-code', model: 'sonnet', phase: 'Recon', schema: RECON }),
() => agent(`${PREAMBLE}
Read-only recon of the CI / lint / coverage / README surface:
- .github/workflows/ci.yaml — quote the pytest invocation (does it pass --cov?),
the ruff steps, the required-check name (ci / ci), any admin-bypass config.
Confirm the Phase 0 AST bundle-consistency test (tests/test_bundle_consistency.py)
is or is not already in the standard run.
- Coverage honesty: confirm there is NO ruff config file today (so PLR/C901 are
unselected) and confirm web_ui/site_extractor lack __init__.py (so a naive
--cov=lambdas SILENTLY skips them). Prove the skip by reading the tree, not by
running.
- scripts/ — list the scripts and confirm they are outside the current lint
scope.
- README.md + docstrings — locate the drift items (findings 33-38 / X-series):
the false "Data is never corrupted" style claims, stale web_ui invocation
contract, missing derived-field/coverage docs, and any test-layout description
that this phase changes. Quote line numbers.
15-25 facts.`,
{ label: 'recon:ci-lint-readme', model: 'haiku', phase: 'Recon', schema: RECON }),
])
const reconOk = recon.filter(Boolean)
const reconBlockers = reconOk.flatMap(r => r.blockers || [])
log(`Recon complete: ${reconOk.length}/4 mappers, ${reconBlockers.length} blockers (all resolved in the main loop — see RESOLUTIONS)`)
// Main-loop resolutions for the first run's recon blockers — verified facts,
// checked directly against WO goldens, live CloudWatch, and the reusable CI
// workflow source. The remaining recon "blockers" were this phase's own work
// items misreported as blockers; none abort the run.
const RESOLUTIONS = `
## MAIN-LOOP RESOLUTIONS (verified AFTER recon flagged blockers — authoritative facts, supersede any recon hedge)
1. WO goldens under parse_float=Decimal: SAFE. All 55 golden files under
lambdas/wo/email_processor/tests/fixtures/expected/ were scanned
programmatically — ZERO float-typed JSON number values exist (decimal-looking
strings appear only inside comment_text STRING values, which parse_float never
touches). The superset load_golden keeping parse_float=Decimal cannot regress WO.
2. malformed_site_code consumers: NONE outside the repo. Live scan of the
deployment account 328440206208 (hosts both po-ingest and WorkorderIngestStack):
0 CloudWatch dashboards, 0 saved Logs Insights query definitions, 18 metric
filters (no match), 146 alarms (no match). seahaven-prod (011934824531) also
clean. Combined with recon's repo-scope grep, the invalid_status reason-code
rename has NO external consumer to break. Still note the rename in the commit
body per the deploy-then-merge outstanding gate.
3. Reusable CI workflow (Sea-Haven-Industries/.github ci-python-sam.yaml
@fd60e4c904, the pinned ref in ci.yaml): inputs.source-dirs threads ONLY into
'ruff check \${{ inputs.source-dirs }}' and 'ruff format --check' — so adding
'scripts' to source-dirs in THIS repo's ci.yaml genuinely puts scripts/ in lint
scope. Tests run as bare 'pytest' with NO args — so --cov and --cov-fail-under
MUST land via pytest.ini addopts (bare pytest picks addopts up); ci.yaml cannot
pass pytest flags. Dependency install is 'pip install pytest' plus
'pip install -r' for EVERY requirements.txt found outside ./.aws-sam — so
pytest-cov (and any other new test dep) must be added to a requirements.txt the
find loop reaches (e.g. a tests/requirements.txt; create it if absent). CI's
ruff is unpinned 'pip install ruff' — the new config must be valid on current ruff.
4. The remaining recon 'blockers' (README drift X1-X8, coverage honesty, absent
ruff config, test_local.py still present) are THIS PHASE'S OWN WORK ITEMS, not
blockers — implement them per the constraints.
`
const pack = reconOk.map(r => `## ${r.summary}\n${r.facts.join('\n')}`).join('\n\n') + '\n\n' + RESOLUTIONS
// -------------------------------------------------------------------- spec
phase('Spec')
const spec = await agent(`${PREAMBLE}
You are the SPEC agent — the single authority that pins every contested
decision BEFORE parallel implementation (parallel leaves cannot see each
other's choices). Using the recon pack below plus your own reads of the actual
files, produce the binding implementation spec:
- rootConftest: the complete new repo-root conftest.py — the dummy AWS env, the
moto BUILTIN_HANDLERS-before-handler-import registration WITH the carried :33
comment, and the single load_lambda_module(pipeline, name) keeping ONE
sys.modules save/restore. State explicitly why the dance does not shrink to
nothing (template_parser bare name) and how the three existing loaders
reconcile (which are deleted, which become thin shims, which re-point).
- supportPackage: tests/support/ — the SUPERSET FakeTable, FakeDynamoResource,
load_email, load_golden. PIN parse_float=Decimal kept in load_golden and
prove WO goldens survive it (recon's non-integer check).
- loaderRewrites: retire the bare-import strategy in _wo_parser_support.py;
rewire both support files + tests/conftest.py onto the new loader; moto
ordering survives; where each fixture lands.
- fileMoves: the git mv set, the stay-at-root set, the test_local.py deletion +
pytest.ini comment removal, per-file import rewrites.
- newScenarios: every new test file path + the scenarios per constraint 5 (a-e).
De-duplicate the reprocess synthetic-event pin against Phase 7 per recon.
- productCodeEdits: the WO handler invalid_status reason-code fix + the
Bedrock-error except-and-reraise metric-wrap with the exact emitted series
(compute by hand: gate-reject emits nothing else; Bedrock-error emits
ai_fallback + bedrock_error ONCE — NO double-count), plus the
malformed_site_code consumer disposition. The PO _validate_new_po_values /
extract_new_po per-rule split (V4 dataclass carries summary_matches/price).
file:line for each; nothing else in these two files changes.
- ruffConfig: the exact config, rule families, ceilings (extract_new_po
C901=35), scripts/ scope, the derived_fields.py per-file-ignore with
justification, and the FULL disposition of every newly-surfaced violation.
- ciChanges: the exact ci.yaml edits — --cov approach (explicit list vs
__init__.py, with reason), fail-under value + placement, AST test + strict
check kept, scripts/ in lint scope, no admin bypass.
- readmeDocs: the drift batch fixes matched to existing style.
- ownership: the DISJOINT 3-agent ownership map with NO shared path, and how
the ruff-config -> product-lint poll dependency resolves.
Recon pack:\n${pack}`,
{ label: 'spec:pin-consolidation', phase: 'Spec', schema: SPEC })
if (!spec) return { aborted: 'spec agent died — rerun workflow', reconBlockers }
const specBlock = `BINDING SPEC (from the spec agent — implement EXACTLY this):\n${JSON.stringify(spec, null, 2)}`
log('Spec pinned: root conftest, support package, loader rewrites, file moves, new scenarios, product-code edits, ruff config, CI changes, README, ownership')
// --------------------------------------------------------------- implement
phase('Implement')
const impl = await parallel([
() => agent(`${PREAMBLE}
YOU OWN: ALL test + support files — the repo-root conftest.py (NEW),
tests/ (including tests/support/ the new package, tests/conftest.py,
tests/test_bundle_consistency.py, the stay-at-root test_parse_raw_email.py /
test_ses_auth.py, and the test_local.py DELETION),
lambdas/po/email_processor/tests/ and lambdas/wo/email_processor/tests/.
You do NOT touch product code, cdk/, ci.yaml, ruff config, pytest.ini, or
README (other agents own those; pytest.ini's test_local.py comment is the
config agent's edit — coordinate only by not touching it).
Task per spec.rootConftest + spec.supportPackage + spec.loaderRewrites +
spec.fileMoves + spec.newScenarios:
1. Write the new repo-root conftest.py with the single load_lambda_module,
carrying the :33 moto comment VERBATIM and keeping ONE sys.modules
save/restore (it does NOT shrink to nothing).
2. Create tests/support/ (superset FakeTable, FakeDynamoResource, load_email,
load_golden with parse_float=Decimal). Retire _wo_parser_support.py's
bare-import strategy; rewire _po_parser_support.py + tests/conftest.py.
3. git mv test_po_merge.py + test_pad_zip.py into the PO tests dir; DELETE
test_local.py; fix all import lines.
4. Add every new scenario suite per constraint 5 (a-e) using ONLY the new
loader + support package. Do NOT edit any existing golden (constraint 9) —
a test that only passes after a golden edit is a bug in the test. WO's one
SES-stamped fixture is the only new fixture allowed.
Run before returning: pytest -q at repo root (all roots green against the OTHER
agents' edits — they land in parallel; the WO metric-wrap test and PO split may
be mid-flight, so poll by re-running up to ~10 min before reporting a blocker),
and confirm goldens are byte-unchanged (git diff --stat on expected/ dirs is
empty).
${specBlock}`,
{ label: 'impl:tests-support', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: lambdas/wo/email_processor/handler.py and
lambdas/po/email_processor/template_parser.py ONLY, PLUS you are the designated
fixer for any NEWLY-SURFACED ruff PLR/C901 violation across product code and
scripts/ (fix-or-noqa-with-justification, NEVER a behavior change). You do NOT
touch tests, cdk/, ci.yaml, README, or derived_fields.py (its violations are
handled via the config agent's per-file-ignore, NOT an edit — constraint 8).
Task per spec.productCodeEdits:
1. WO handler: apply the invalid_status reason-code fix (only after confirming
spec captured the malformed_site_code consumer disposition) and the
Bedrock-error except-and-RERAISE metric-wrap. The wrap emits
ai_fallback + bedrock_error ONCE on a Bedrock error and NOTHING extra on a
gate reject — do the hand series-count in your summary. Handler event/return
contract UNCHANGED.
2. PO template_parser: the _validate_new_po_values / extract_new_po per-rule
split; the V4 anchor-frame dataclass carries summary_matches/price for V13.
3. The ruff config lands in a sibling agent's file. POLL for it (re-check every
~60s up to ~10 min); once present, run 'ruff check <config> .' and resolve
EVERY newly-surfaced violation in the files you own (and scripts/) — fix or
noqa with a written justification. If a violation lands in a file you do NOT
own and is not derived_fields.py, report it as a blocker for the fix loop.
Run before returning: ruff check + ruff format --check on the files you touched,
and a targeted pytest of the WO Bedrock-fallback + PO validation-gate suites
(poll for the test agent's new files up to ~10 min).
${specBlock}`,
{ label: 'impl:product-code', model: 'opus', phase: 'Implement', schema: IMPL }),
() => agent(`${PREAMBLE}
YOU OWN: the ruff config file (pyproject.toml or ruff.toml — spec picks which),
pytest.ini, .github/workflows/ci.yaml, and README.md ONLY. You do NOT touch
tests or product code.
Task per spec.ruffConfig + spec.ciChanges + spec.readmeDocs:
1. Create the ruff config ENABLING C901/PLR with the pinned ceilings
(extract_new_po C901=35), scripts/ in scope, and the derived_fields.py
per-file-ignore with a justification comment (constraint 8). Write this
FIRST so the product-code agent can poll for it.
2. pytest.ini: remove the test_local.py exclusion comment (the file is being
deleted by the test agent); keep the three testpaths roots.
3. ci.yaml: add --cov with the spec's approach (explicit module list OR
__init__.py note), a fail-under, keep the AST bundle-consistency test in the
standard run, add scripts/ to the lint scope, keep the strict 'ci / ci'
required check, no admin bypass.
4. README + docstring drift batch (findings 33-38): the false "Data is never
corrupted" claims, stale web_ui invocation contract, the new test layout
(single loader, tests/support/, moved files), coverage/ruff-floor note.
Run before returning: ruff check <your new config> . to confirm the config is
valid and to enumerate what it surfaces (report the list for the product-code
agent), and a yaml lint / dry parse of ci.yaml. Do NOT run the full suite (the
test agent owns that gate).
${specBlock}`,
{ label: 'impl:config-ci-readme', model: 'sonnet', phase: 'Implement', schema: IMPL }),
])
const implOk = impl.filter(Boolean)
const implBlockers = implOk.flatMap(r => r.blockers || [])
log(`Implement complete: ${implOk.length}/3 agents, blockers: ${implBlockers.length}`)
// ---------------------------------------------------- verify + fix loop
const EXPECTED_SCOPE = [
'conftest.py',
'tests/',
'lambdas/po/email_processor/tests/',
'lambdas/wo/email_processor/tests/',
'lambdas/wo/email_processor/handler.py',
'lambdas/po/email_processor/template_parser.py',
'pyproject.toml',
'ruff.toml',
'pytest.ini',
'.github/workflows/ci.yaml',
'README.md',
]
const mechanicalPrompt = `${PREAMBLE}
Independent re-verification — trust nothing self-reported. Run ALL gates,
quoting failures verbatim:
1. pytest -q at repo root — all three roots collected and GREEN under the NEW
single loader. Then prove the loader actually runs: pytest lambdas/po alone
and pytest lambdas/wo alone each collect and pass (the new root conftest
must load for a standalone run — the whole reason it moved to repo root).
2. GOLDENS UNCHANGED: git diff ${BASE}...HEAD -- '**/expected/*.json' and the
.eml corpora must be empty (constraint 9). Any golden edit = FAIL.
3. ruff check . && ruff format --check . under the NEW config, WITH scripts/ in
scope. Zero violations (every surfaced one fixed or noqa'd-with-justification;
derived_fields.py handled via per-file-ignore, its file byte-unchanged:
git diff ${BASE}...HEAD -- lambdas/po/email_processor/derived_fields.py empty).
4. cd cdk && npx cdk synth po-ingest -q && npx cdk synth workorder-ingest -q
(artifact-id selectors) — both synth clean (no CDK change this phase, but a
broken import would surface here).
5. COVERAGE HONESTY: run the CI --cov invocation and CONFIRM the coverage report
lists lambdas/po/web_ui, lambdas/wo/web_ui and lambdas/po/site_extractor with
NON-zero, NON-omitted lines (prove they are no longer silently skipped for a
missing __init__.py). Confirm the fail-under is present and would actually
fail if tripped (has teeth).
6. WO NO-DOUBLE-COUNT PIN: run the specific test that computes the WO
Bedrock-error emitted series by hand and assert it passes; then git grep for
the reason-code rename and confirm no dashboard/metric-filter consumer of
'malformed_site_code' was left dangling.
7. UNTOUCHABLES (each must output NOTHING): git diff ${BASE}...HEAD for
derived_fields.py, both ses_auth, both validate_ai_fallback gates,
lambdas/shared/, all cdk/*.py, site_extractor. The ONLY product diffs allowed
are lambdas/wo/email_processor/handler.py and
lambdas/po/email_processor/template_parser.py.
8. FILE MOVES + DELETION: test_po_merge.py + test_pad_zip.py are gone from
tests/ and present under lambdas/po/email_processor/tests/; test_local.py is
deleted; test_parse_raw_email.py + test_ses_auth.py still at root and green.
9. AST bundle-consistency test present in the standard run and green.
10. git status --porcelain scope check: every modified/added/deleted path under
${EXPECTED_SCOPE.join(', ')} plus tests/support/ (untracked
.coverage/.claude/package/ tolerated).
passed=true only if all green. YOU MAY NOT edit files.`
const lenses = [
{ key: 'loader-integrity', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — loader-integrity lens. ONE loader replaced three; attack
it. (1) Prove there is exactly ONE load path now and that the moto
BUILTIN_HANDLERS registration still happens BEFORE any handler import in EVERY
entry order (session start AND a standalone 'pytest lambdas/wo' run) — if a
handler's module-level boto3 client can be created before moto registers, the
moto-backed suites silently hit real AWS. (2) sys.modules isolation for
template_parser: run TWO cross-pipeline tests back to back (a PO golden then a
WO golden, and the reverse) and prove the save/restore still binds each handler
to its OWN template_parser — the dance did NOT shrink to nothing. (3) The file
moves (test_po_merge/test_pad_zip into the PO dir) must not break collection or
silently drop a test — count tests before/after. (4) Did retiring
_wo_parser_support.py's bare import leave any stale 'import handler' /
sys.path.insert that could re-introduce the collision? (5) load_golden's
parse_float=Decimal survived for PO (exact money) and did not corrupt WO.
confirmed=true only with file:line or a reproduced-failure.` },
{ key: 'wo-metric-contract', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — WO-metric-contract lens. The WO Bedrock-error metric-wrap
is the one behavior change; prove it does NOT break wo_stack's alarm contract.
Read the wrapped handler diff (git diff ${BASE}) and wo_stack's fallback-rate
MathExpression + the "a rejected email emits nothing else" alarm. By HAND,
enumerate the emitted metric series for: (a) a normal template parse, (b) a
gate-rejected email, (c) a Bedrock error (Throttling / missing-content /
empty-content / non-JSON). Prove NO series is double-counted — specifically that
the except-and-reraise emits ai_fallback + bedrock_error EXACTLY ONCE on a
Bedrock error and that a gate-rejected email still emits ONLY its rejected
datapoint and nothing else (a naive reorder would double-count — confirm this
was NOT done). Then the reason-code fix: confirm invalid_status no longer
collides with malformed_site_code and that no dashboard/metric-filter consumer
of the old code is left dangling. Confirm the handler event/return contract is
unchanged (no signature drift). confirmed=true only with the hand-computed
series tables as evidence.` },
{ key: 'coverage-honesty', prompt: `${PREAMBLE}
ADVERSARIAL REVIEW — coverage-honesty lens. The headline coverage historically
overstated reality because --cov=lambdas silently skipped web_ui/site_extractor
(missing __init__.py). Prove that is FIXED: run the new CI --cov invocation and
show lambdas/po/web_ui, lambdas/wo/web_ui, lambdas/po/site_extractor each appear
in the report with real measured lines (not omitted, not 0-of-0). Prove the
fail-under has TEETH — temporarily lower a threshold or add a trivially-uncovered
line on a scratch copy and show the job would fail (then discard). Prove the new
web_ui suites exercise the REAL auth module (fail-closed on unset ARN + on a
Secrets-Manager exception, 401 WITHOUT a table scan, non-ASCII token) and not a
mock that would pass with the gate deleted. Finally, confirm the ruff C901/PLR
config genuinely bites (extract_new_po at C901=35 is enforced, the previously
inert noqas are now live and justified, scripts/ is in scope, and
derived_fields.py is excluded via config not an edit). confirmed=true only with
command output as evidence.` },
]
let round = 0
let checks = null
let confirmed = []
while (round < 3) {
phase('Verify')
const results = await parallel([
() => agent(mechanicalPrompt, { label: `verify:mechanical-r${round}`, model: 'sonnet', phase: 'Verify', schema: CHECKS }),
...lenses.map(l => () =>
agent(l.prompt, { label: `verify:${l.key}-r${round}`, phase: 'Verify', schema: FINDINGS })),
])
checks = results[0]
confirmed = results.slice(1).filter(Boolean)
.flatMap(r => r.findings || [])
.filter(f => f.confirmed && f.severity !== 'low')
const green = checks && checks.passed
log(`Verify round ${round}: mechanical ${checks && checks.passed ? 'GREEN' : 'RED'}, confirmed findings: ${confirmed.length}`)
if (green && confirmed.length === 0) break
round += 1
if (round >= 3) break
phase('Fix')
await agent(`${PREAMBLE}
You are the fix agent — you may edit files under: ${EXPECTED_SCOPE.join(', ')}
and tests/support/. Fix EVERY item below minimally; the binding spec and the
10 pinned constraints still hold (a finding that conflicts with a constraint is
reported, not "fixed" — the constraint wins, esp. constraint 9's untouchable
goldens + derived_fields freeze, and constraint 7's no-double-count metric
wrap: do NOT convert it to a naive reorder to silence a finding). Re-run the
specific failing gate/test per fix.
MECHANICAL:\n${checks ? checks.details : '(agent died — rerun all gates)'}
CONFIRMED FINDINGS:\n${JSON.stringify(confirmed, null, 2)}
${specBlock}`,
{ label: `fix:round-${round}`, model: 'opus', phase: 'Fix', schema: IMPL })
}
const verifyClean = checks && checks.passed && confirmed.length === 0
if (!verifyClean) {
return {
status: 'NEEDS ATTENTION — verify not clean after 3 rounds; branch left uncommitted',
branch: BRANCH,
mechanical: checks,
unresolvedFindings: confirmed,
implBlockers,
reconBlockers,
spec,
}
}
// ----------------------------------------------------------------- package
phase('Package')
const commit = await agent(`${PREAMBLE.replace('do NOT commit, ', '')}
YOU are the commit agent:
1. Read ~/Documents/repositories/seahaven/engineering-handbook/commit-messages.md
and follow it exactly.
2. git add only paths under: ${EXPECTED_SCOPE.join(', ')}, tests/support/, and
.claude/workflows/phase-8-test-consolidation.js. NOT .coverage, NOT package/.
Verify the staged set with git status — the test_local.py DELETION and the
git mv of test_po_merge.py / test_pad_zip.py MUST be staged too.
3. ONE commit; write the message to /tmp/phase8-commit-msg.txt and use
git commit -F /tmp/phase8-commit-msg.txt (backticks in -m get eaten by zsh).
Suggested subject:
"test: consolidate test roots — one repo-root loader, shared support package, missing-scenario suites, enforced ruff/coverage floor (refactor phase 8)"
Body: the single repo-root conftest loader + moto-before-handler carry, the
tests/support/ superset (parse_float=Decimal kept), the file moves +
test_local.py deletion, the new scenario suites (Bedrock transport, SES-auth
seam, web_ui both, WO merge, small pins), the WO Bedrock-error metric-wrap
(except-and-reraise, no double-count) + invalid_status reason-code fix, the
PO _validate_new_po_values split, the enforced ruff C901/PLR config, and the
CI --cov + fail-under floor. NO AI attribution / Co-Authored-By lines.
4. Do NOT push. Return commit sha + shortstat in summary.`,
{ label: 'package:commit', model: 'sonnet', phase: 'Package', schema: IMPL })
return {
status: 'BUILT — committed locally, NOT pushed',
branch: BRANCH,
base: BASE,
commit: commit ? commit.summary : 'commit agent died — commit manually',
spec: { rootConftest: spec.rootConftest, productCodeEdits: spec.productCodeEdits, ruffConfig: spec.ruffConfig, ciChanges: spec.ciChanges, ownership: spec.ownership },
implementation: implOk.map(r => r.summary),
filesChanged: implOk.flatMap(r => r.filesChanged),
verifyRounds: round + 1,
blockers: implBlockers.concat(reconBlockers),
outstandingGates: [
'/sh-security-review RECOMMENDED before push — the WO Bedrock-error metric-wrap is a behavior change on the untrusted-input processing path, and web_ui auth is the mandatory-review surface (here only TESTS are added for it). Flag it for the WO metric change specifically; re-run after any post-review fix.',
'cross-family cross_review.py NOT required — no IAM/policy change and no handler-signature change (the WO metric-wrap keeps the event/return contract). Opt-in only if judgment says so.',
'deploy-then-merge for the WO handler metric/reason-code change: deploy from branch, smoke green, one live WO email, verify the ParseOutcome series still matches the alarm contract (NO double-count), and grep the dashboards for malformed_site_code BEFORE the reason-code rename ships.',
'LAST PHASE: the new suite / coverage fail-under / enforced ruff C901/PLR config become the new CI floor — do not weaken any existing gate to land a later change.',
],
}