From a6bb1adfbcf6b7bcbcbc5738659d4c384ea4da2f Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Fri, 1 May 2026 19:52:07 -0400 Subject: [PATCH] Add reusable chunk scraper, LLM prompt, README; clean up one-off scripts - Add scrape-fill-chunk.mjs: reusable chunk-based Payee Central scraper with resume support and concurrent instance capability - Add prompts/po-email-parser.md: LLM prompt for Coupa PO email parsing with site code extraction patterns, skip list, and trade classification - Add README.md with architecture docs, DynamoDB schema, and script index - Track output/site-state-extra-mapping.json (392 facility codes) - Remove scrape-payee.mjs and scrape-payee-notfound.mjs (superseded by chunk scraper) --- .gitignore | 3 +- README.md | 57 +++ output/site-state-extra-mapping.json | 408 ++++++++++++++++++++++ prompts/po-email-parser.md | 181 ++++++++++ scrape-payee.mjs => scrape-fill-chunk.mjs | 92 +++-- scrape-payee-notfound.mjs | 170 --------- 6 files changed, 690 insertions(+), 221 deletions(-) create mode 100644 README.md create mode 100644 output/site-state-extra-mapping.json create mode 100644 prompts/po-email-parser.md rename scrape-payee.mjs => scrape-fill-chunk.mjs (56%) delete mode 100644 scrape-payee-notfound.mjs diff --git a/.gitignore b/.gitignore index 44cf079..e51e6b7 100644 --- a/.gitignore +++ b/.gitignore @@ -1,6 +1,7 @@ node_modules/ .env -output/ +output/* +!output/site-state-extra-mapping.json chrome-profile*/ cookies.json .DS_Store diff --git a/README.md b/README.md new file mode 100644 index 0000000..1d7c23c --- /dev/null +++ b/README.md @@ -0,0 +1,57 @@ +# amazon-po-parser + +Tools for ingesting, scraping, and enriching Amazon purchase order data from Coupa emails and Payee Central. + +## Architecture + +- **Data source**: Coupa PO notification emails → parsed into structured JSON +- **Enrichment**: Playwright scrapers hit Amazon Payee Central to resolve site codes, ship-to addresses, and line items +- **Storage**: DynamoDB `purchase-orders` table (us-east-1) is the single source of truth +- **Mapping**: `output/site-state-extra-mapping.json` — curated Amazon facility code → US state mapping (392 codes) + +## Scripts + +| Script | Purpose | +|---|---| +| `parse-mbox.py` | Parse Coupa PO emails from MBOX export into structured JSON | +| `scrape.mjs` | Single-browser Payee Central scraper | +| `scrape-payee-chunk.mjs` | Chunk-based concurrent scraper (run N instances in parallel) | +| `scrape-fill-chunk.mjs` | Lightweight scraper for backfilling site codes and ship-to addresses | +| `gen-po-list.mjs` | Generate PO lists from invoice spreadsheets | +| `monitor.mjs` | Monitor scraper progress across chunks | + +## DynamoDB Schema + +Table: `purchase-orders` | Key: `po_number` (string) + +| Field | Type | Source | +|---|---|---| +| `po_number` | string | Email/Coupa | +| `site_code` | string | Payee Central scrape | +| `state` | string | Derived from site_code mapping | +| `ship_to_raw` | string | Payee Central scrape | +| `line_items` | list | Email parse + Payee Central | +| `trade` | string | Classified from descriptions | +| `trades` | list | All trades for multi-line POs | +| `fiscal_year` | string | Order date year | +| `coupa_category` | string | Coupa commodity category | +| `source_system` | string | "coupa" | +| `data_source` | string | "email+payee_scrape" | +| `invoice_total` | string | Downstream invoice matching | +| `invoice_count` | number | Downstream invoice matching | +| `order_date` | string | Email parse | +| `po_status` | string | Email parse | +| `payment_terms` | string | Email parse | +| `requisition_number` | string | Email parse | + +## Prompt + +`prompts/po-email-parser.md` contains the LLM prompt for parsing incoming Coupa PO emails. It includes site code extraction patterns, a skip list for non-site codes, trade classification rules, and ship-to address parsing guidance. + +## Setup + +```bash +npm install # playwright +``` + +Requires a Chrome profile logged into Amazon Payee Central at `./chrome-profile/`. diff --git a/output/site-state-extra-mapping.json b/output/site-state-extra-mapping.json new file mode 100644 index 0000000..8d2dadf --- /dev/null +++ b/output/site-state-extra-mapping.json @@ -0,0 +1,408 @@ +{ + "1001": "IL", + "11710": "WA", + "120": "IN", + "1214": "TX", + "13079": "CO", + "13537": "WA", + "13905": "MD", + "14248": "NE", + "150": "MA", + "1521": "NY", + "1788": "CA", + "18867": "FL", + "2081": "SC", + "220": "ND", + "2302": "GA", + "231": "NY", + "2702": "MN", + "28869": "AL", + "3121": "PA", + "3201": "MD", + "322": "NC", + "330": "MA", + "3334": "CA", + "3623": "OK", + "4100": "VA", + "4101": "PA", + "4200": "MD", + "4221": "TX", + "4500": "TX", + "4525": "NC", + "455": "CA", + "4769": "FL", + "480": "CO", + "4825": "TX", + "4925": "GA", + "4976": "CA", + "500": "AZ", + "5119": "CA", + "550": "TX", + "5501": "MD", + "555N": "AZ", + "5650": "CA", + "5802": "NV", + "5915": "MS", + "6250": "CA", + "6715": "FL", + "717": "CA", + "7211": "NY", + "750": "CA", + "761": "OK", + "850": "MS", + "8720": "MT", + "8817": "CA", + "923": "CA", + "9269": "CA", + "DCS6": "CO", + "DML6": "WI", + "DNA6": "TN", + "DOM3": "NE", + "DTB4": "MA", + "HBN9": "TN", + "HDS2": "IA", + "HHO3": "TX", + "HJX2": "FL", + "HLO9": "IN", + "HMI3": "FL", + "MCI4": "MO", + "ORF5": "VA", + "POK1": "OK", + "RSR": "TX", + "SAZ3": "AZ", + "SFI1": "FL", + "SID1": "ID", + "SKY2": "KY", + "SNL1": "TN", + "SSC3": "SC", + "UIL2": "IL", + "WBM4": "AL", + "WBY1": "TX", + "WCO8": "CO", + "WFB1": "AK", + "WFG1": "NM", + "WFL6": "FL", + "WGC8": "KS", + "WGR1": "GA", + "WGR2": "GA", + "WGR3": "GA", + "WID8": "ID", + "WIL5": "IL", + "WIN5": "IN", + "WIO2": "IA", + "WIO4": "IA", + "WIO9": "IA", + "WKN3": "AR", + "WKS3": "KS", + "WLN3": "LA", + "WMD1": "MD", + "WME5": "ME", + "WMN3": "MN", + "WMO4": "MO", + "WMS2": "MS", + "WMT2": "MT", + "WMT4": "MT", + "WNB2": "NE", + "WNB3": "NE", + "WNB4": "NE", + "WNC3": "NC", + "WNC4": "NC", + "WNC8": "NC", + "WNC9": "NC", + "WND4": "ND", + "WND6": "ND", + "WNG1": "TX", + "WNM2": "NM", + "WNM4": "NM", + "WNV3": "NV", + "WOH2": "OH", + "WOL1": "CO", + "WOR3": "OR", + "WOR4": "OR", + "WOR9": "OR", + "WPY1": "PA", + "WRT3": "TX", + "WSC2": "SC", + "WSC7": "SC", + "WSD2": "SD", + "WSM9": "MO", + "WTX8": "TX", + "WTX9": "TX", + "WUT1": "UT", + "WUT3": "UT", + "WUT4": "UT", + "WWG1": "WA", + "WWG2": "WA", + "WWG4": "WA", + "WWG6": "WA", + "WWG9": "WA", + "WWV9": "WV", + "WWY1": "WY", + "WWY4": "WY", + "WZN4": "AZ", + "WZN6": "AZ", + "ZAT7": "GA", + "ZDL7": "TX", + "HHO9": "TX", + "PPP1": "PA", + "PCL1": "OH", + "ZBL1": "WA", + "WIS2": "WI", + "APC1": "CA", + "PPO1": "IN", + "HBA2": "MD", + "HRD2": "NC", + "HBD1": "CT", + "HLA2": "CA", + "HBN2": "TN", + "HAT2": "GA", + "HIN3": "IN", + "HDE2": "CO", + "HDY1": "OH", + "HCE2": "OH", + "HPH4": "DE", + "HNY2": "NY", + "HNY5": "NJ", + "HAL1": "NY", + "HVB2": "VA", + "HBM3": "AL", + "HBO2": "MA", + "HNY3": "NY", + "HLU2": "MO", + "HMK4": "WI", + "HCT2": "TN", + "DYO1": "NY", + "DFM5": "FL", + "AMH7": "TX", + "KLAL": "FL", + "HLO3": "MA", + "HRS1": "FL", + "HMO2": "AL", + "HMY1": "LA", + "SNJ1": "NJ", + "SYR1": "NY", + "STW1": "MI", + "ORF2": "VA", + "WWI6": "WI", + "SJA1": "FL", + "UGA4": "GA", + "DJR5": "NJ", + "DNJ7": "NJ", + "WNY3": "NY", + "XEV1": "IN", + "WNC5": "NC", + "UNC3": "NC", + "WNC6": "NC", + "WQQ1": "NM", + "WTX5": "TX", + "WID1": "ID", + "WTX1": "TX", + "XSF2": "AR", + "WDE1": "DE", + "ULA6": "CA", + "WIS1": "WI", + "HCO1": "CO", + "WKY3": "KY", + "WLC1": "NM", + "WNM3": "NM", + "SCA1": "CA", + "HDT3": "MI", + "WTH1": "CA", + "WTX4": "TX", + "WMO2": "MO", + "WIN2": "IN", + "WMT1": "MT", + "7930": "FL", + "HSD2": "CA", + "HCH5": "IL", + "WEE1": "PA", + "ZAT3": "GA", + "USF2": "CA", + "UCA5": "CA", + "SAZ2": "AZ", + "WIN1": "IN", + "XAQ1": "NM", + "UNY2": "NY", + "WIL4": "IL", + "WID3": "ID", + "HMW1": "IL", + "PCO2": "CO", + "LDJ5": "NY", + "WWI3": "WI", + "PHL4": "PA", + "WMO1": "MO", + "HPH2": "PA", + "WTX2": "TX", + "HPT1": "PA", + "WIL1": "IL", + "UAZ1": "AZ", + "WSD1": "SD", + "WCH2": "VA", + "WWI2": "IA", + "USF1": "CA", + "HKX1": "TN", + "USD1": "CA", + "HSA1": "GA", + "WSP1": "AR", + "WMN2": "MN", + "HSM1": "CA", + "UWI2": "WI", + "XCV8": "VA", + "WTN1": "VA", + "WFL2": "FL", + "WMS1": "MS", + "WMO3": "MO", + "9475": "OR", + "WND1": "ND", + "170": "NJ", + "XTE3": "CA", + "UTX8": "TX", + "AZA2": "AZ", + "WMI1": "MI", + "USF4": "CA", + "135": "NJ", + "HBO1": "MA", + "SWF2": "PA", + "POA3": "CA", + "UFL5": "FL", + "UOH5": "OH", + "WOO1": "PA", + "HPX3": "AZ", + "HDC3": "PA", + "HSL1": "UT", + "UOR2": "OR", + "ZSE2": "WA", + "UOH4": "OH", + "UNC2": "NC", + "HGR1": "MI", + "XWI4": "OK", + "6426": "VA", + "HDA3": "TX", + "WQQ2": "NM", + "1700": "VA", + "UFL4": "FL", + "WWS1": "PA", + "2900": "FL", + "WWI4": "WI", + "GAT1": "GA", + "WTX3": "TX", + "ZST3": "WA", + "XSA1": "MI", + "UMA4": "MA", + "UNV3": "NV", + "14311": "CA", + "WML1": "FL", + "XTE2": "CA", + "XMS4": "MS", + "XCV7": "VA", + "LSB1": "CA", + "1217": "TX", + "6573": "PA", + "WFS1": "MN", + "WMT3": "MT", + "BUR2": "CA", + "SWI1": "WI", + "WWY8": "WY", + "353": "OR", + "WFL3": "FL", + "HFA2": "CA", + "3100": "AL", + "PPH3": "PA", + "WKS1": "KS", + "UVA4": "VA", + "HLV1": "NV", + "9607": "TX", + "WOH1": "OH", + "PMI2": "FL", + "2635": "FL", + "390": "CA", + "3881": "CA", + "DEP1": "IL", + "10133": "TX", + "HOM1": "NE", + "QPI8": "NC", + "XCV3": "VA", + "100": "NV", + "1400": "NJ", + "DAV1": "CA", + "19800": "VA", + "WET1": "MS", + "5300": "TN", + "299": "FL", + "IRV1": "CA", + "SBA1": "CA", + "4501": "VA", + "POR2": "FL", + "XCV5": "VA", + "LBB1": "TX", + "2006": "MO", + "OSU1": "OH", + "HSY1": "NY", + "UTX7": "TX", + "CHI3": "IL", + "255": "TX", + "UVA5": "VA", + "HSE1": "MD", + "UCO1": "CO", + "SCA7": "CA", + "HOK2": "OK", + "XSA2": "MI", + "HNY1": "NY", + "3736": "PA", + "7395": "FL", + "LSD5": "CA", + "XJO1": "OK", + "2897": "MI", + "HMI2": "FL", + "HSF9": "CA", + "WZN1": "AZ", + "SNY1": "NJ", + "DYR3": "NY", + "301": "TN", + "HMB1": "FL", + "2001": "CA", + "UNY5": "NY", + "HRO1": "NY", + "SFL8": "FL", + "3250": "CA", + "WNY1": "NY", + "SMO1": "MO", + "HAU1": "TX", + "6610": "NC", + "471": "NJ", + "7469": "FL", + "UNY4": "NY", + "2951": "NV", + "SAT6": "TX", + "300": "MD", + "2100": "MI", + "CFL1": "FL", + "495": "NY", + "DPD8": "OR", + "2153": "CA", + "UFL6": "FL", + "323": "AZ", + "BDL6": "CT", + "2203": "MD", + "XWI5": "OK", + "2040": "CA", + "QIW2": "IA", + "11041": "TX", + "2847": "CA", + "QAV9": "WV", + "TPA6": "FL", + "DLT6": "NC", + "AND8": "ND", + "WUS5": "TX", + "HLX1": "CA", + "DBK4": "NY", + "PVD2": "RI", + "DJZ5": "NJ", + "JFK2": "NY", + "DNK5": "NJ", + "SWF4": "FL", + "DCY9": "CT", + "DJZ4": "NJ", + "AWY2": "WY", + "DYY4": "NY" +} \ No newline at end of file diff --git a/prompts/po-email-parser.md b/prompts/po-email-parser.md new file mode 100644 index 0000000..e071fa3 --- /dev/null +++ b/prompts/po-email-parser.md @@ -0,0 +1,181 @@ +You are an email parser for a purchase order ingest pipeline. +The emails are Coupa procurement platform notifications containing purchase order +data from Amazon. + +Analyze the following email and extract structured data. Return ONLY valid JSON with these fields: + +```json +{ + "email_type": "new_po" | "revision" | "cancellation", + "po_number": "string or null", + "po_status": "string or null", + "source_system": "coupa", + "submitted_by": "string or null", + "on_behalf_of": "string or null", + "order_date": "string or null", + "revision_date": "string or null", + "last_opened": "string or null", + "acknowledged_at": "string or null", + "payment_terms": "string or null", + "requisition_number": "string or null", + "department": "string or null", + "view_order_url": "URL string or null", + "supplier": { + "name": "string or null" + }, + "site_code": "string or null", + "ship_to": { + "name": "string or null", + "address": "string or null", + "street": "string or null", + "city": "string or null", + "state": "string or null", + "zip": "string or null", + "location_code": "string or null", + "attn": "string or null" + }, + "total_amount": 0.0, + "currency": "USD", + "fiscal_year": "string or null", + "trade": "string or null", + "coupa_category": "string or null", + "line_items": [ + { + "description": "string", + "amount": 0.0, + "currency": "USD", + "need_by": "date string or null", + "category": "string or null", + "account_code": "string or null", + "period": "string or null", + "quantity": "string or null", + "unit": "string or null", + "price": "string or null" + } + ] +} +``` + +## email_type detection + +- "new_po": email announces a new purchase order being issued +- "revision": email announces a revised/updated purchase order (look for "revised" in subject or body) +- "cancellation": email announces a PO has been cancelled + +## PO number + +Extract from the email subject or body. Format is a prefix + hyphen + digits: +- "2D-18206023", "FK-21088051", "B187-17955555" + +## site_code extraction + +The site code is the Amazon facility code — a 3-5 character alphanumeric code identifying +the delivery site. Check these locations in order: + +1. Ship-to name in parentheses: "Amazon.com Services LLC (KLAL)" → KLAL +2. Ship-to name after dash: "Amazon.com Services LLC - SNY5" → SNY5 +3. Ship-to ATTN line with dash or en-dash: "ATTN: Wagon Wheel DS Station –WKY3" → WKY3 +4. Ship-to ATTN line directly: "Attn: HJX1" → HJX1 +5. Ship-to name IS the code: if the name is just "DBU2" or similar, use it +6. Line item description prefix: "DYO1 - Sea Haven Ind - Plumbing Repairs" → DYO1 +7. Line item description in brackets: "[HMK4] Assemble 3 Wire Security Cages" → HMK4 + +**Not site codes — do not extract these as site_code:** +- RME (Amazon Reliability Maintenance Engineering department) +- BBM (Coupa description format tag) +- JLL (Jones Lang LaSalle — facilities management vendor) +- PARAG, ERIK (vendor/person names) +- Industry acronyms: HVAC, LED, PVC, ADA, OSHA, EMR, BMS, DDC, MRO, NTE, EST + +If the only candidate matches this skip list, set site_code to null. + +## Ship-to address parsing + +Parse the full address into separate fields. Be aware of these common issues: +- State abbreviation may be missing entirely (e.g., "Tucson, 85704" with no state) +- Zip codes may lack leading zeros (e.g., "MA 2149" should be zip "02149", "NJ 7001" should be "07001") +- City names may be misspelled (e.g., "Charoltte" for Charlotte) — extract as-is, do not correct +- Format varies: "City, ST - ZIP", "City, ST ZIP", "City, ZIP" (no state) + +If state cannot be determined from the address, set ship_to.state to null. + +## fiscal_year + +The calendar year the work covers. Determine from: +1. The order_date year (primary source) +2. Need-by dates on line items +3. Year in line item descriptions (e.g., "HVB2 - 2025 - Plumbing PM" → "2025") + +Use the 4-digit year string (e.g., "2025"). + +## trade classification + +Classify the primary trade from line item descriptions. Use the FIRST match in priority order: + +**Plumbing - PM**: "plumbing pm", "plumbing preventative", "plumbing maintenance", + or BBM format: "Plumbing - Backflow", "Plumbing - Water Heater - Install/Repair" + +**Plumbing - Reactive**: "plumbing" with: "reactive", "emergency", "repair", "clog", + "unclog", "leak", "flood", "sewer", "drain", "grease trap", "jetter", "water line", + "toilet", "faucet", "urinal", "pipe" + +**Electrical**: "electrical", "lighting", "ballast", "outlet", "circuit", "panel", + "generator", "transformer", "conduit" (but NOT if "dock door" context) + +**HVAC**: "hvac", "heating", "cooling", "air conditioning", "RTU", "AHU", "VAV", + "refrigerant", "thermostat", "ductwork" + +**Dock Doors**: "dock door", "dock leveler", "dock plate", "dock seal", "dock bumper" + +**Doors**: "door", "overhead door", "roll-up", "automatic door", "access door" + (only if not matched by Dock Doors above) + +**Signage**: "sign", "banner", "wayfinding", "marquee", "directional" + +**Carpentry**: "carpentry", "cabinet", "millwork", "trim", "shelving", "framing" + +**Fencing/Gates**: "fence", "fencing", "gate", "bollard" (not "dock gate") + +**Conveyance/MHE**: "conveyor", "MHE", "material handling", "sortation" + +**Painting**: "paint", "painting", "primer", "coating", "touch-up" + +**Flooring**: "floor", "tile", "carpet", "epoxy", "polishing" + +**Janitorial**: "janitorial", "cleaning", "custodial", "pressure wash", "power wash" + +**Fire/Life Safety**: "fire", "sprinkler", "extinguisher", "fire alarm", "suppression" + +**Landscaping/Yard**: "landscape", "lawn", "tree", "yard", "mowing", "irrigation" + +**Roofing**: "roof", "roofing", "gutter", "downspout" + +**Security/Locksmith**: "lock", "key", "access control", "camera", "security", "CCTV" + +**Snow Removal**: "snow", "ice", "salt", "de-ice", "plow" + +**PO Uplift**: description is exactly or primarily "PO Uplift" + +**General Building - Emergency**: "EMER" prefix, or "emergency" in a general building context + +**General Building - Handyman**: BBM format "General Building - General Building Technician" + +**General Building - Project**: BBM format "General Building - General Building Project" + +**General Building**: any remaining facility maintenance work + +If a PO has multiple line items with different trades, set "trade" to the primary +(non-uplift, non-materials) trade. If genuinely mixed, use the trade of the highest-value line item. + +## coupa_category + +The Coupa commodity/category field if present in the email (e.g., "Maintenance - Facilities", +"Plumbing Equipment & Materials"). This is Coupa's own classification, not the trade field. + +## General rules + +- Extract all line items with descriptions, amounts, and metadata +- "quantity", "unit" (e.g., "EACH", "HR"), and "price" (unit price) should be extracted when present +- total_amount should be the numeric total in USD +- If a field is not present in the email, set it to null +- Do NOT invent or infer data that is not explicitly in the email diff --git a/scrape-payee.mjs b/scrape-fill-chunk.mjs similarity index 56% rename from scrape-payee.mjs rename to scrape-fill-chunk.mjs index c0b47c8..4b68b17 100644 --- a/scrape-payee.mjs +++ b/scrape-fill-chunk.mjs @@ -1,22 +1,30 @@ import { chromium } from "playwright"; import { readFileSync, writeFileSync, existsSync } from "fs"; +import { execSync } from "child_process"; -const PO_LIST = "output/unknown-pos.json"; -const OUTPUT_FILE = "output/payee-results.json"; -const PROGRESS_FILE = "output/payee-progress.json"; +const chunkId = process.argv[2]; +if (!chunkId && chunkId !== "0") { + console.error("Usage: node scrape-fill-chunk.mjs "); + process.exit(1); +} + +const PO_LIST = `output/fill-chunk${chunkId}.json`; +const OUTPUT_FILE = `output/fill-chunk${chunkId}-results.json`; +const PROGRESS_FILE = `output/fill-chunk${chunkId}-progress.json`; +const PROFILE_DIR = `chrome-profile-fill-${chunkId}`; const DELAY_MS = 800; const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8")); -console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`); +console.log(`[Chunk ${chunkId}] Loaded ${allPOs.length} POs`); let completed = {}; if (existsSync(PROGRESS_FILE)) { completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8")); - console.log(`Resuming: ${Object.keys(completed).length} already scraped`); + console.log(`[Chunk ${chunkId}] Resuming: ${Object.keys(completed).length} already done`); } const remaining = allPOs.filter((po) => !completed[po]); -console.log(`Remaining: ${remaining.length} POs\n`); +console.log(`[Chunk ${chunkId}] Remaining: ${remaining.length} POs\n`); if (remaining.length === 0) { console.log("All POs already scraped!"); @@ -27,8 +35,14 @@ function poUrl(po) { return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`; } -const userDataDir = new URL("./chrome-profile", import.meta.url).pathname; -const context = await chromium.launchPersistentContext(userDataDir, { +const profilePath = new URL(`./${PROFILE_DIR}`, import.meta.url).pathname; +const mainProfile = new URL("./chrome-profile", import.meta.url).pathname; +if (!existsSync(profilePath)) { + console.log(`Copying chrome profile to ${PROFILE_DIR}...`); + execSync(`cp -R "${mainProfile}" "${profilePath}"`); +} + +const context = await chromium.launchPersistentContext(profilePath, { headless: false, channel: "chrome", args: ["--disable-blink-features=AutomationControlled"], @@ -37,11 +51,11 @@ const context = await chromium.launchPersistentContext(userDataDir, { const page = context.pages()[0] || (await context.newPage()); -console.log("Opening Payee Central..."); +console.log(`[Chunk ${chunkId}] Opening Payee Central...`); await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 }); if (page.url().includes("signin") || page.url().includes("/ap/")) { - console.log("\n🔐 Login required. Please sign in in the browser window..."); + console.log(`\n[Chunk ${chunkId}] Login required. Please sign in...`); await page.waitForURL( (u) => { const s = u.toString(); @@ -49,7 +63,7 @@ if (page.url().includes("signin") || page.url().includes("/ap/")) { }, { timeout: 180000 } ); - console.log("✅ Logged in!\n"); + console.log(`[Chunk ${chunkId}] Logged in!\n`); } let results = []; @@ -67,7 +81,7 @@ for (const po of remaining) { await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 }); if (page.url().includes("signin") || page.url().includes("/ap/")) { - console.log("\n🔐 Session expired. Please log in again..."); + console.log(`\n[Chunk ${chunkId}] Session expired. Please log in again...`); await page.waitForURL( (u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"), { timeout: 180000 } @@ -78,46 +92,29 @@ for (const po of remaining) { await page.waitForTimeout(500); const bodyText = await page.innerText("body"); - // Extract Ship To with site code const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/); const shipTo = shipToMatch ? shipToMatch[1].trim() : null; let siteCode = null; if (shipTo) { - const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/); - if (m) siteCode = m[1]; - if (!siteCode) { - const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/); - if (m2) siteCode = m2[1]; + for (const re of [ + /\(([A-Z0-9]{3,5})\)/, + /(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/, + /LLC\s+([A-Z0-9]{3,5})\b/, + /^([A-Z0-9]{3,5})\s*-\s*Amazon/m, + /Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/, + ]) { + const m = shipTo.match(re); + if (m) { siteCode = m[1]; break; } } } - // Extract line items from table - const lineItems = await page.$$eval( - "table tbody tr", - (rows) => - rows.map((row) => { - const cells = row.querySelectorAll("td"); - if (cells.length < 7) return null; - return { - description: cells[1]?.textContent?.trim() || null, - need_by: cells[2]?.textContent?.trim() || null, - quantity: cells[3]?.textContent?.trim() || null, - unit: cells[4]?.textContent?.trim() || null, - price: cells[5]?.textContent?.trim() || null, - amount: cells[6]?.textContent?.trim() || null, - }; - }).filter(Boolean) - ).catch(() => []); - - const result = { + results.push({ po_number: po, site_code: siteCode, ship_to_raw: shipTo, - line_items: lineItems, - }; + }); - results.push(result); completed[po] = true; scraped++; consecutiveErrors = 0; @@ -129,15 +126,13 @@ for (const po of remaining) { const etaSec = eta % 60; process.stdout.write( - `\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s ` + `\r[C${chunkId}] [${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s ` ); - if (scraped % 10 === 0) { - writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2)); - writeFileSync(PROGRESS_FILE, JSON.stringify(completed)); - } + writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2)); + writeFileSync(PROGRESS_FILE, JSON.stringify(completed)); } catch (err) { - console.error(`\n [${po}] Error: ${err.message}`); + console.error(`\n[C${chunkId}] [${po}] Error: ${err.message}`); errors++; consecutiveErrors++; if (consecutiveErrors > 10) { @@ -152,8 +147,5 @@ for (const po of remaining) { writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2)); writeFileSync(PROGRESS_FILE, JSON.stringify(completed)); -console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`); -console.log(`Total in output: ${results.length} POs`); -console.log(`Results saved to ${OUTPUT_FILE}`); - +console.log(`\n\n[Chunk ${chunkId}] Done! Scraped: ${scraped}, Errors: ${errors}`); await context.close(); diff --git a/scrape-payee-notfound.mjs b/scrape-payee-notfound.mjs deleted file mode 100644 index 7fe79f8..0000000 --- a/scrape-payee-notfound.mjs +++ /dev/null @@ -1,170 +0,0 @@ -import { chromium } from "playwright"; -import { readFileSync, writeFileSync, existsSync } from "fs"; - -const PO_LIST = "output/not-found-pos.json"; -const OUTPUT_FILE = "output/notfound-payee-results.json"; -const PROGRESS_FILE = "output/notfound-payee-progress.json"; -const DELAY_MS = 800; - -const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8")); -console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`); - -let completed = {}; -if (existsSync(PROGRESS_FILE)) { - completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8")); - console.log(`Resuming: ${Object.keys(completed).length} already scraped`); -} - -const remaining = allPOs.filter((po) => !completed[po]); -console.log(`Remaining: ${remaining.length} POs\n`); - -if (remaining.length === 0) { - console.log("All POs already scraped!"); - process.exit(0); -} - -function poUrl(po) { - return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`; -} - -const userDataDir = new URL("./chrome-profile", import.meta.url).pathname; -const context = await chromium.launchPersistentContext(userDataDir, { - headless: false, - channel: "chrome", - args: ["--disable-blink-features=AutomationControlled"], - ignoreDefaultArgs: ["--enable-automation"], -}); - -const page = context.pages()[0] || (await context.newPage()); - -console.log("Opening Payee Central..."); -await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 }); - -if (page.url().includes("signin") || page.url().includes("/ap/")) { - console.log("\n🔐 Login required. Please sign in in the browser window..."); - await page.waitForURL( - (u) => { - const s = u.toString(); - return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/"); - }, - { timeout: 180000 } - ); - console.log("✅ Logged in!\n"); -} - -let results = []; -if (existsSync(OUTPUT_FILE)) { - results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8")); -} - -let scraped = 0; -let errors = 0; -let consecutiveErrors = 0; -const startTime = Date.now(); - -for (const po of remaining) { - try { - await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 }); - - if (page.url().includes("signin") || page.url().includes("/ap/")) { - console.log("\n🔐 Session expired. Please log in again..."); - await page.waitForURL( - (u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"), - { timeout: 180000 } - ); - await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 }); - } - - await page.waitForTimeout(500); - const bodyText = await page.innerText("body"); - - const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/); - const shipTo = shipToMatch ? shipToMatch[1].trim() : null; - - let siteCode = null; - if (shipTo) { - const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/); - if (m) siteCode = m[1]; - if (!siteCode) { - const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/); - if (m2) siteCode = m2[1]; - } - if (!siteCode) { - const m3 = shipTo.match(/LLC\s+([A-Z0-9]{3,5})\b/); - if (m3) siteCode = m3[1]; - } - if (!siteCode) { - const m4 = shipTo.match(/^([A-Z0-9]{3,5})\s*-\s*Amazon/m); - if (m4) siteCode = m4[1]; - } - if (!siteCode) { - const m5 = shipTo.match(/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/); - if (m5) siteCode = m5[1]; - } - } - - const lineItems = await page.$$eval( - "table tbody tr", - (rows) => - rows.map((row) => { - const cells = row.querySelectorAll("td"); - if (cells.length < 7) return null; - return { - description: cells[1]?.textContent?.trim() || null, - need_by: cells[2]?.textContent?.trim() || null, - quantity: cells[3]?.textContent?.trim() || null, - unit: cells[4]?.textContent?.trim() || null, - price: cells[5]?.textContent?.trim() || null, - amount: cells[6]?.textContent?.trim() || null, - }; - }).filter(Boolean) - ).catch(() => []); - - const result = { - po_number: po, - site_code: siteCode, - ship_to_raw: shipTo, - line_items: lineItems, - }; - - results.push(result); - completed[po] = true; - scraped++; - consecutiveErrors = 0; - - const elapsed = (Date.now() - startTime) / 1000; - const rate = scraped / elapsed; - const eta = Math.round((remaining.length - scraped) / rate); - const etaMin = Math.floor(eta / 60); - const etaSec = eta % 60; - - const desc0 = lineItems[0]?.description?.substring(0, 40) || "no items"; - process.stdout.write( - `\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ${desc0} | ETA: ${etaMin}m${etaSec}s ` - ); - - if (scraped % 10 === 0) { - writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2)); - writeFileSync(PROGRESS_FILE, JSON.stringify(completed)); - } - } catch (err) { - console.error(`\n [${po}] Error: ${err.message}`); - errors++; - consecutiveErrors++; - if (consecutiveErrors > 10) { - console.error("Too many consecutive errors, stopping."); - break; - } - } - - await page.waitForTimeout(DELAY_MS); -} - -writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2)); -writeFileSync(PROGRESS_FILE, JSON.stringify(completed)); - -console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`); -console.log(`Total in output: ${results.length} POs`); -console.log(`Results saved to ${OUTPUT_FILE}`); - -await context.close();