Add reusable chunk scraper, LLM prompt, README; clean up one-off scripts

- Add scrape-fill-chunk.mjs: reusable chunk-based Payee Central scraper
  with resume support and concurrent instance capability
- Add prompts/po-email-parser.md: LLM prompt for Coupa PO email parsing
  with site code extraction patterns, skip list, and trade classification
- Add README.md with architecture docs, DynamoDB schema, and script index
- Track output/site-state-extra-mapping.json (392 facility codes)
- Remove scrape-payee.mjs and scrape-payee-notfound.mjs (superseded by
  chunk scraper)
This commit is contained in:
Adam Moussa 2026-05-01 19:52:07 -04:00
parent 17442e053a
commit a6bb1adfbc
6 changed files with 690 additions and 221 deletions

3
.gitignore vendored
View file

@ -1,6 +1,7 @@
node_modules/
.env
output/
output/*
!output/site-state-extra-mapping.json
chrome-profile*/
cookies.json
.DS_Store

57
README.md Normal file
View file

@ -0,0 +1,57 @@
# amazon-po-parser
Tools for ingesting, scraping, and enriching Amazon purchase order data from Coupa emails and Payee Central.
## Architecture
- **Data source**: Coupa PO notification emails → parsed into structured JSON
- **Enrichment**: Playwright scrapers hit Amazon Payee Central to resolve site codes, ship-to addresses, and line items
- **Storage**: DynamoDB `purchase-orders` table (us-east-1) is the single source of truth
- **Mapping**: `output/site-state-extra-mapping.json` — curated Amazon facility code → US state mapping (392 codes)
## Scripts
| Script | Purpose |
|---|---|
| `parse-mbox.py` | Parse Coupa PO emails from MBOX export into structured JSON |
| `scrape.mjs` | Single-browser Payee Central scraper |
| `scrape-payee-chunk.mjs` | Chunk-based concurrent scraper (run N instances in parallel) |
| `scrape-fill-chunk.mjs` | Lightweight scraper for backfilling site codes and ship-to addresses |
| `gen-po-list.mjs` | Generate PO lists from invoice spreadsheets |
| `monitor.mjs` | Monitor scraper progress across chunks |
## DynamoDB Schema
Table: `purchase-orders` | Key: `po_number` (string)
| Field | Type | Source |
|---|---|---|
| `po_number` | string | Email/Coupa |
| `site_code` | string | Payee Central scrape |
| `state` | string | Derived from site_code mapping |
| `ship_to_raw` | string | Payee Central scrape |
| `line_items` | list | Email parse + Payee Central |
| `trade` | string | Classified from descriptions |
| `trades` | list | All trades for multi-line POs |
| `fiscal_year` | string | Order date year |
| `coupa_category` | string | Coupa commodity category |
| `source_system` | string | "coupa" |
| `data_source` | string | "email+payee_scrape" |
| `invoice_total` | string | Downstream invoice matching |
| `invoice_count` | number | Downstream invoice matching |
| `order_date` | string | Email parse |
| `po_status` | string | Email parse |
| `payment_terms` | string | Email parse |
| `requisition_number` | string | Email parse |
## Prompt
`prompts/po-email-parser.md` contains the LLM prompt for parsing incoming Coupa PO emails. It includes site code extraction patterns, a skip list for non-site codes, trade classification rules, and ship-to address parsing guidance.
## Setup
```bash
npm install # playwright
```
Requires a Chrome profile logged into Amazon Payee Central at `./chrome-profile/`.

View file

@ -0,0 +1,408 @@
{
"1001": "IL",
"11710": "WA",
"120": "IN",
"1214": "TX",
"13079": "CO",
"13537": "WA",
"13905": "MD",
"14248": "NE",
"150": "MA",
"1521": "NY",
"1788": "CA",
"18867": "FL",
"2081": "SC",
"220": "ND",
"2302": "GA",
"231": "NY",
"2702": "MN",
"28869": "AL",
"3121": "PA",
"3201": "MD",
"322": "NC",
"330": "MA",
"3334": "CA",
"3623": "OK",
"4100": "VA",
"4101": "PA",
"4200": "MD",
"4221": "TX",
"4500": "TX",
"4525": "NC",
"455": "CA",
"4769": "FL",
"480": "CO",
"4825": "TX",
"4925": "GA",
"4976": "CA",
"500": "AZ",
"5119": "CA",
"550": "TX",
"5501": "MD",
"555N": "AZ",
"5650": "CA",
"5802": "NV",
"5915": "MS",
"6250": "CA",
"6715": "FL",
"717": "CA",
"7211": "NY",
"750": "CA",
"761": "OK",
"850": "MS",
"8720": "MT",
"8817": "CA",
"923": "CA",
"9269": "CA",
"DCS6": "CO",
"DML6": "WI",
"DNA6": "TN",
"DOM3": "NE",
"DTB4": "MA",
"HBN9": "TN",
"HDS2": "IA",
"HHO3": "TX",
"HJX2": "FL",
"HLO9": "IN",
"HMI3": "FL",
"MCI4": "MO",
"ORF5": "VA",
"POK1": "OK",
"RSR": "TX",
"SAZ3": "AZ",
"SFI1": "FL",
"SID1": "ID",
"SKY2": "KY",
"SNL1": "TN",
"SSC3": "SC",
"UIL2": "IL",
"WBM4": "AL",
"WBY1": "TX",
"WCO8": "CO",
"WFB1": "AK",
"WFG1": "NM",
"WFL6": "FL",
"WGC8": "KS",
"WGR1": "GA",
"WGR2": "GA",
"WGR3": "GA",
"WID8": "ID",
"WIL5": "IL",
"WIN5": "IN",
"WIO2": "IA",
"WIO4": "IA",
"WIO9": "IA",
"WKN3": "AR",
"WKS3": "KS",
"WLN3": "LA",
"WMD1": "MD",
"WME5": "ME",
"WMN3": "MN",
"WMO4": "MO",
"WMS2": "MS",
"WMT2": "MT",
"WMT4": "MT",
"WNB2": "NE",
"WNB3": "NE",
"WNB4": "NE",
"WNC3": "NC",
"WNC4": "NC",
"WNC8": "NC",
"WNC9": "NC",
"WND4": "ND",
"WND6": "ND",
"WNG1": "TX",
"WNM2": "NM",
"WNM4": "NM",
"WNV3": "NV",
"WOH2": "OH",
"WOL1": "CO",
"WOR3": "OR",
"WOR4": "OR",
"WOR9": "OR",
"WPY1": "PA",
"WRT3": "TX",
"WSC2": "SC",
"WSC7": "SC",
"WSD2": "SD",
"WSM9": "MO",
"WTX8": "TX",
"WTX9": "TX",
"WUT1": "UT",
"WUT3": "UT",
"WUT4": "UT",
"WWG1": "WA",
"WWG2": "WA",
"WWG4": "WA",
"WWG6": "WA",
"WWG9": "WA",
"WWV9": "WV",
"WWY1": "WY",
"WWY4": "WY",
"WZN4": "AZ",
"WZN6": "AZ",
"ZAT7": "GA",
"ZDL7": "TX",
"HHO9": "TX",
"PPP1": "PA",
"PCL1": "OH",
"ZBL1": "WA",
"WIS2": "WI",
"APC1": "CA",
"PPO1": "IN",
"HBA2": "MD",
"HRD2": "NC",
"HBD1": "CT",
"HLA2": "CA",
"HBN2": "TN",
"HAT2": "GA",
"HIN3": "IN",
"HDE2": "CO",
"HDY1": "OH",
"HCE2": "OH",
"HPH4": "DE",
"HNY2": "NY",
"HNY5": "NJ",
"HAL1": "NY",
"HVB2": "VA",
"HBM3": "AL",
"HBO2": "MA",
"HNY3": "NY",
"HLU2": "MO",
"HMK4": "WI",
"HCT2": "TN",
"DYO1": "NY",
"DFM5": "FL",
"AMH7": "TX",
"KLAL": "FL",
"HLO3": "MA",
"HRS1": "FL",
"HMO2": "AL",
"HMY1": "LA",
"SNJ1": "NJ",
"SYR1": "NY",
"STW1": "MI",
"ORF2": "VA",
"WWI6": "WI",
"SJA1": "FL",
"UGA4": "GA",
"DJR5": "NJ",
"DNJ7": "NJ",
"WNY3": "NY",
"XEV1": "IN",
"WNC5": "NC",
"UNC3": "NC",
"WNC6": "NC",
"WQQ1": "NM",
"WTX5": "TX",
"WID1": "ID",
"WTX1": "TX",
"XSF2": "AR",
"WDE1": "DE",
"ULA6": "CA",
"WIS1": "WI",
"HCO1": "CO",
"WKY3": "KY",
"WLC1": "NM",
"WNM3": "NM",
"SCA1": "CA",
"HDT3": "MI",
"WTH1": "CA",
"WTX4": "TX",
"WMO2": "MO",
"WIN2": "IN",
"WMT1": "MT",
"7930": "FL",
"HSD2": "CA",
"HCH5": "IL",
"WEE1": "PA",
"ZAT3": "GA",
"USF2": "CA",
"UCA5": "CA",
"SAZ2": "AZ",
"WIN1": "IN",
"XAQ1": "NM",
"UNY2": "NY",
"WIL4": "IL",
"WID3": "ID",
"HMW1": "IL",
"PCO2": "CO",
"LDJ5": "NY",
"WWI3": "WI",
"PHL4": "PA",
"WMO1": "MO",
"HPH2": "PA",
"WTX2": "TX",
"HPT1": "PA",
"WIL1": "IL",
"UAZ1": "AZ",
"WSD1": "SD",
"WCH2": "VA",
"WWI2": "IA",
"USF1": "CA",
"HKX1": "TN",
"USD1": "CA",
"HSA1": "GA",
"WSP1": "AR",
"WMN2": "MN",
"HSM1": "CA",
"UWI2": "WI",
"XCV8": "VA",
"WTN1": "VA",
"WFL2": "FL",
"WMS1": "MS",
"WMO3": "MO",
"9475": "OR",
"WND1": "ND",
"170": "NJ",
"XTE3": "CA",
"UTX8": "TX",
"AZA2": "AZ",
"WMI1": "MI",
"USF4": "CA",
"135": "NJ",
"HBO1": "MA",
"SWF2": "PA",
"POA3": "CA",
"UFL5": "FL",
"UOH5": "OH",
"WOO1": "PA",
"HPX3": "AZ",
"HDC3": "PA",
"HSL1": "UT",
"UOR2": "OR",
"ZSE2": "WA",
"UOH4": "OH",
"UNC2": "NC",
"HGR1": "MI",
"XWI4": "OK",
"6426": "VA",
"HDA3": "TX",
"WQQ2": "NM",
"1700": "VA",
"UFL4": "FL",
"WWS1": "PA",
"2900": "FL",
"WWI4": "WI",
"GAT1": "GA",
"WTX3": "TX",
"ZST3": "WA",
"XSA1": "MI",
"UMA4": "MA",
"UNV3": "NV",
"14311": "CA",
"WML1": "FL",
"XTE2": "CA",
"XMS4": "MS",
"XCV7": "VA",
"LSB1": "CA",
"1217": "TX",
"6573": "PA",
"WFS1": "MN",
"WMT3": "MT",
"BUR2": "CA",
"SWI1": "WI",
"WWY8": "WY",
"353": "OR",
"WFL3": "FL",
"HFA2": "CA",
"3100": "AL",
"PPH3": "PA",
"WKS1": "KS",
"UVA4": "VA",
"HLV1": "NV",
"9607": "TX",
"WOH1": "OH",
"PMI2": "FL",
"2635": "FL",
"390": "CA",
"3881": "CA",
"DEP1": "IL",
"10133": "TX",
"HOM1": "NE",
"QPI8": "NC",
"XCV3": "VA",
"100": "NV",
"1400": "NJ",
"DAV1": "CA",
"19800": "VA",
"WET1": "MS",
"5300": "TN",
"299": "FL",
"IRV1": "CA",
"SBA1": "CA",
"4501": "VA",
"POR2": "FL",
"XCV5": "VA",
"LBB1": "TX",
"2006": "MO",
"OSU1": "OH",
"HSY1": "NY",
"UTX7": "TX",
"CHI3": "IL",
"255": "TX",
"UVA5": "VA",
"HSE1": "MD",
"UCO1": "CO",
"SCA7": "CA",
"HOK2": "OK",
"XSA2": "MI",
"HNY1": "NY",
"3736": "PA",
"7395": "FL",
"LSD5": "CA",
"XJO1": "OK",
"2897": "MI",
"HMI2": "FL",
"HSF9": "CA",
"WZN1": "AZ",
"SNY1": "NJ",
"DYR3": "NY",
"301": "TN",
"HMB1": "FL",
"2001": "CA",
"UNY5": "NY",
"HRO1": "NY",
"SFL8": "FL",
"3250": "CA",
"WNY1": "NY",
"SMO1": "MO",
"HAU1": "TX",
"6610": "NC",
"471": "NJ",
"7469": "FL",
"UNY4": "NY",
"2951": "NV",
"SAT6": "TX",
"300": "MD",
"2100": "MI",
"CFL1": "FL",
"495": "NY",
"DPD8": "OR",
"2153": "CA",
"UFL6": "FL",
"323": "AZ",
"BDL6": "CT",
"2203": "MD",
"XWI5": "OK",
"2040": "CA",
"QIW2": "IA",
"11041": "TX",
"2847": "CA",
"QAV9": "WV",
"TPA6": "FL",
"DLT6": "NC",
"AND8": "ND",
"WUS5": "TX",
"HLX1": "CA",
"DBK4": "NY",
"PVD2": "RI",
"DJZ5": "NJ",
"JFK2": "NY",
"DNK5": "NJ",
"SWF4": "FL",
"DCY9": "CT",
"DJZ4": "NJ",
"AWY2": "WY",
"DYY4": "NY"
}

181
prompts/po-email-parser.md Normal file
View file

@ -0,0 +1,181 @@
You are an email parser for a purchase order ingest pipeline.
The emails are Coupa procurement platform notifications containing purchase order
data from Amazon.
Analyze the following email and extract structured data. Return ONLY valid JSON with these fields:
```json
{
"email_type": "new_po" | "revision" | "cancellation",
"po_number": "string or null",
"po_status": "string or null",
"source_system": "coupa",
"submitted_by": "string or null",
"on_behalf_of": "string or null",
"order_date": "string or null",
"revision_date": "string or null",
"last_opened": "string or null",
"acknowledged_at": "string or null",
"payment_terms": "string or null",
"requisition_number": "string or null",
"department": "string or null",
"view_order_url": "URL string or null",
"supplier": {
"name": "string or null"
},
"site_code": "string or null",
"ship_to": {
"name": "string or null",
"address": "string or null",
"street": "string or null",
"city": "string or null",
"state": "string or null",
"zip": "string or null",
"location_code": "string or null",
"attn": "string or null"
},
"total_amount": 0.0,
"currency": "USD",
"fiscal_year": "string or null",
"trade": "string or null",
"coupa_category": "string or null",
"line_items": [
{
"description": "string",
"amount": 0.0,
"currency": "USD",
"need_by": "date string or null",
"category": "string or null",
"account_code": "string or null",
"period": "string or null",
"quantity": "string or null",
"unit": "string or null",
"price": "string or null"
}
]
}
```
## email_type detection
- "new_po": email announces a new purchase order being issued
- "revision": email announces a revised/updated purchase order (look for "revised" in subject or body)
- "cancellation": email announces a PO has been cancelled
## PO number
Extract from the email subject or body. Format is a prefix + hyphen + digits:
- "2D-18206023", "FK-21088051", "B187-17955555"
## site_code extraction
The site code is the Amazon facility code — a 3-5 character alphanumeric code identifying
the delivery site. Check these locations in order:
1. Ship-to name in parentheses: "Amazon.com Services LLC (KLAL)" → KLAL
2. Ship-to name after dash: "Amazon.com Services LLC - SNY5" → SNY5
3. Ship-to ATTN line with dash or en-dash: "ATTN: Wagon Wheel DS Station –WKY3" → WKY3
4. Ship-to ATTN line directly: "Attn: HJX1" → HJX1
5. Ship-to name IS the code: if the name is just "DBU2" or similar, use it
6. Line item description prefix: "DYO1 - Sea Haven Ind - Plumbing Repairs" → DYO1
7. Line item description in brackets: "[HMK4] Assemble 3 Wire Security Cages" → HMK4
**Not site codes — do not extract these as site_code:**
- RME (Amazon Reliability Maintenance Engineering department)
- BBM (Coupa description format tag)
- JLL (Jones Lang LaSalle — facilities management vendor)
- PARAG, ERIK (vendor/person names)
- Industry acronyms: HVAC, LED, PVC, ADA, OSHA, EMR, BMS, DDC, MRO, NTE, EST
If the only candidate matches this skip list, set site_code to null.
## Ship-to address parsing
Parse the full address into separate fields. Be aware of these common issues:
- State abbreviation may be missing entirely (e.g., "Tucson, 85704" with no state)
- Zip codes may lack leading zeros (e.g., "MA 2149" should be zip "02149", "NJ 7001" should be "07001")
- City names may be misspelled (e.g., "Charoltte" for Charlotte) — extract as-is, do not correct
- Format varies: "City, ST - ZIP", "City, ST ZIP", "City, ZIP" (no state)
If state cannot be determined from the address, set ship_to.state to null.
## fiscal_year
The calendar year the work covers. Determine from:
1. The order_date year (primary source)
2. Need-by dates on line items
3. Year in line item descriptions (e.g., "HVB2 - 2025 - Plumbing PM" → "2025")
Use the 4-digit year string (e.g., "2025").
## trade classification
Classify the primary trade from line item descriptions. Use the FIRST match in priority order:
**Plumbing - PM**: "plumbing pm", "plumbing preventative", "plumbing maintenance",
or BBM format: "Plumbing - Backflow", "Plumbing - Water Heater - Install/Repair"
**Plumbing - Reactive**: "plumbing" with: "reactive", "emergency", "repair", "clog",
"unclog", "leak", "flood", "sewer", "drain", "grease trap", "jetter", "water line",
"toilet", "faucet", "urinal", "pipe"
**Electrical**: "electrical", "lighting", "ballast", "outlet", "circuit", "panel",
"generator", "transformer", "conduit" (but NOT if "dock door" context)
**HVAC**: "hvac", "heating", "cooling", "air conditioning", "RTU", "AHU", "VAV",
"refrigerant", "thermostat", "ductwork"
**Dock Doors**: "dock door", "dock leveler", "dock plate", "dock seal", "dock bumper"
**Doors**: "door", "overhead door", "roll-up", "automatic door", "access door"
(only if not matched by Dock Doors above)
**Signage**: "sign", "banner", "wayfinding", "marquee", "directional"
**Carpentry**: "carpentry", "cabinet", "millwork", "trim", "shelving", "framing"
**Fencing/Gates**: "fence", "fencing", "gate", "bollard" (not "dock gate")
**Conveyance/MHE**: "conveyor", "MHE", "material handling", "sortation"
**Painting**: "paint", "painting", "primer", "coating", "touch-up"
**Flooring**: "floor", "tile", "carpet", "epoxy", "polishing"
**Janitorial**: "janitorial", "cleaning", "custodial", "pressure wash", "power wash"
**Fire/Life Safety**: "fire", "sprinkler", "extinguisher", "fire alarm", "suppression"
**Landscaping/Yard**: "landscape", "lawn", "tree", "yard", "mowing", "irrigation"
**Roofing**: "roof", "roofing", "gutter", "downspout"
**Security/Locksmith**: "lock", "key", "access control", "camera", "security", "CCTV"
**Snow Removal**: "snow", "ice", "salt", "de-ice", "plow"
**PO Uplift**: description is exactly or primarily "PO Uplift"
**General Building - Emergency**: "EMER" prefix, or "emergency" in a general building context
**General Building - Handyman**: BBM format "General Building - General Building Technician"
**General Building - Project**: BBM format "General Building - General Building Project"
**General Building**: any remaining facility maintenance work
If a PO has multiple line items with different trades, set "trade" to the primary
(non-uplift, non-materials) trade. If genuinely mixed, use the trade of the highest-value line item.
## coupa_category
The Coupa commodity/category field if present in the email (e.g., "Maintenance - Facilities",
"Plumbing Equipment & Materials"). This is Coupa's own classification, not the trade field.
## General rules
- Extract all line items with descriptions, amounts, and metadata
- "quantity", "unit" (e.g., "EACH", "HR"), and "price" (unit price) should be extracted when present
- total_amount should be the numeric total in USD
- If a field is not present in the email, set it to null
- Do NOT invent or infer data that is not explicitly in the email

View file

@ -1,22 +1,30 @@
import { chromium } from "playwright";
import { readFileSync, writeFileSync, existsSync } from "fs";
import { execSync } from "child_process";
const PO_LIST = "output/unknown-pos.json";
const OUTPUT_FILE = "output/payee-results.json";
const PROGRESS_FILE = "output/payee-progress.json";
const chunkId = process.argv[2];
if (!chunkId && chunkId !== "0") {
console.error("Usage: node scrape-fill-chunk.mjs <chunk-id>");
process.exit(1);
}
const PO_LIST = `output/fill-chunk${chunkId}.json`;
const OUTPUT_FILE = `output/fill-chunk${chunkId}-results.json`;
const PROGRESS_FILE = `output/fill-chunk${chunkId}-progress.json`;
const PROFILE_DIR = `chrome-profile-fill-${chunkId}`;
const DELAY_MS = 800;
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`);
console.log(`[Chunk ${chunkId}] Loaded ${allPOs.length} POs`);
let completed = {};
if (existsSync(PROGRESS_FILE)) {
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
console.log(`[Chunk ${chunkId}] Resuming: ${Object.keys(completed).length} already done`);
}
const remaining = allPOs.filter((po) => !completed[po]);
console.log(`Remaining: ${remaining.length} POs\n`);
console.log(`[Chunk ${chunkId}] Remaining: ${remaining.length} POs\n`);
if (remaining.length === 0) {
console.log("All POs already scraped!");
@ -27,8 +35,14 @@ function poUrl(po) {
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
}
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
const context = await chromium.launchPersistentContext(userDataDir, {
const profilePath = new URL(`./${PROFILE_DIR}`, import.meta.url).pathname;
const mainProfile = new URL("./chrome-profile", import.meta.url).pathname;
if (!existsSync(profilePath)) {
console.log(`Copying chrome profile to ${PROFILE_DIR}...`);
execSync(`cp -R "${mainProfile}" "${profilePath}"`);
}
const context = await chromium.launchPersistentContext(profilePath, {
headless: false,
channel: "chrome",
args: ["--disable-blink-features=AutomationControlled"],
@ -37,11 +51,11 @@ const context = await chromium.launchPersistentContext(userDataDir, {
const page = context.pages()[0] || (await context.newPage());
console.log("Opening Payee Central...");
console.log(`[Chunk ${chunkId}] Opening Payee Central...`);
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Login required. Please sign in in the browser window...");
console.log(`\n[Chunk ${chunkId}] Login required. Please sign in...`);
await page.waitForURL(
(u) => {
const s = u.toString();
@ -49,7 +63,7 @@ if (page.url().includes("signin") || page.url().includes("/ap/")) {
},
{ timeout: 180000 }
);
console.log("✅ Logged in!\n");
console.log(`[Chunk ${chunkId}] Logged in!\n`);
}
let results = [];
@ -67,7 +81,7 @@ for (const po of remaining) {
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Session expired. Please log in again...");
console.log(`\n[Chunk ${chunkId}] Session expired. Please log in again...`);
await page.waitForURL(
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
{ timeout: 180000 }
@ -78,46 +92,29 @@ for (const po of remaining) {
await page.waitForTimeout(500);
const bodyText = await page.innerText("body");
// Extract Ship To with site code
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
let siteCode = null;
if (shipTo) {
const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/);
if (m) siteCode = m[1];
if (!siteCode) {
const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/);
if (m2) siteCode = m2[1];
for (const re of [
/\(([A-Z0-9]{3,5})\)/,
/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/,
/LLC\s+([A-Z0-9]{3,5})\b/,
/^([A-Z0-9]{3,5})\s*-\s*Amazon/m,
/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/,
]) {
const m = shipTo.match(re);
if (m) { siteCode = m[1]; break; }
}
}
// Extract line items from table
const lineItems = await page.$$eval(
"table tbody tr",
(rows) =>
rows.map((row) => {
const cells = row.querySelectorAll("td");
if (cells.length < 7) return null;
return {
description: cells[1]?.textContent?.trim() || null,
need_by: cells[2]?.textContent?.trim() || null,
quantity: cells[3]?.textContent?.trim() || null,
unit: cells[4]?.textContent?.trim() || null,
price: cells[5]?.textContent?.trim() || null,
amount: cells[6]?.textContent?.trim() || null,
};
}).filter(Boolean)
).catch(() => []);
const result = {
results.push({
po_number: po,
site_code: siteCode,
ship_to_raw: shipTo,
line_items: lineItems,
};
});
results.push(result);
completed[po] = true;
scraped++;
consecutiveErrors = 0;
@ -129,15 +126,13 @@ for (const po of remaining) {
const etaSec = eta % 60;
process.stdout.write(
`\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s `
`\r[C${chunkId}] [${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s `
);
if (scraped % 10 === 0) {
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
}
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
} catch (err) {
console.error(`\n [${po}] Error: ${err.message}`);
console.error(`\n[C${chunkId}] [${po}] Error: ${err.message}`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
@ -152,8 +147,5 @@ for (const po of remaining) {
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
console.log(`Total in output: ${results.length} POs`);
console.log(`Results saved to ${OUTPUT_FILE}`);
console.log(`\n\n[Chunk ${chunkId}] Done! Scraped: ${scraped}, Errors: ${errors}`);
await context.close();

View file

@ -1,170 +0,0 @@
import { chromium } from "playwright";
import { readFileSync, writeFileSync, existsSync } from "fs";
const PO_LIST = "output/not-found-pos.json";
const OUTPUT_FILE = "output/notfound-payee-results.json";
const PROGRESS_FILE = "output/notfound-payee-progress.json";
const DELAY_MS = 800;
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`);
let completed = {};
if (existsSync(PROGRESS_FILE)) {
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
}
const remaining = allPOs.filter((po) => !completed[po]);
console.log(`Remaining: ${remaining.length} POs\n`);
if (remaining.length === 0) {
console.log("All POs already scraped!");
process.exit(0);
}
function poUrl(po) {
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
}
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
const context = await chromium.launchPersistentContext(userDataDir, {
headless: false,
channel: "chrome",
args: ["--disable-blink-features=AutomationControlled"],
ignoreDefaultArgs: ["--enable-automation"],
});
const page = context.pages()[0] || (await context.newPage());
console.log("Opening Payee Central...");
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Login required. Please sign in in the browser window...");
await page.waitForURL(
(u) => {
const s = u.toString();
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
},
{ timeout: 180000 }
);
console.log("✅ Logged in!\n");
}
let results = [];
if (existsSync(OUTPUT_FILE)) {
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
}
let scraped = 0;
let errors = 0;
let consecutiveErrors = 0;
const startTime = Date.now();
for (const po of remaining) {
try {
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Session expired. Please log in again...");
await page.waitForURL(
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
{ timeout: 180000 }
);
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
}
await page.waitForTimeout(500);
const bodyText = await page.innerText("body");
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
let siteCode = null;
if (shipTo) {
const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/);
if (m) siteCode = m[1];
if (!siteCode) {
const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/);
if (m2) siteCode = m2[1];
}
if (!siteCode) {
const m3 = shipTo.match(/LLC\s+([A-Z0-9]{3,5})\b/);
if (m3) siteCode = m3[1];
}
if (!siteCode) {
const m4 = shipTo.match(/^([A-Z0-9]{3,5})\s*-\s*Amazon/m);
if (m4) siteCode = m4[1];
}
if (!siteCode) {
const m5 = shipTo.match(/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/);
if (m5) siteCode = m5[1];
}
}
const lineItems = await page.$$eval(
"table tbody tr",
(rows) =>
rows.map((row) => {
const cells = row.querySelectorAll("td");
if (cells.length < 7) return null;
return {
description: cells[1]?.textContent?.trim() || null,
need_by: cells[2]?.textContent?.trim() || null,
quantity: cells[3]?.textContent?.trim() || null,
unit: cells[4]?.textContent?.trim() || null,
price: cells[5]?.textContent?.trim() || null,
amount: cells[6]?.textContent?.trim() || null,
};
}).filter(Boolean)
).catch(() => []);
const result = {
po_number: po,
site_code: siteCode,
ship_to_raw: shipTo,
line_items: lineItems,
};
results.push(result);
completed[po] = true;
scraped++;
consecutiveErrors = 0;
const elapsed = (Date.now() - startTime) / 1000;
const rate = scraped / elapsed;
const eta = Math.round((remaining.length - scraped) / rate);
const etaMin = Math.floor(eta / 60);
const etaSec = eta % 60;
const desc0 = lineItems[0]?.description?.substring(0, 40) || "no items";
process.stdout.write(
`\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ${desc0} | ETA: ${etaMin}m${etaSec}s `
);
if (scraped % 10 === 0) {
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
}
} catch (err) {
console.error(`\n [${po}] Error: ${err.message}`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
console.error("Too many consecutive errors, stopping.");
break;
}
}
await page.waitForTimeout(DELAY_MS);
}
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
console.log(`Total in output: ${results.length} POs`);
console.log(`Results saved to ${OUTPUT_FILE}`);
await context.close();