Initial commit: Coupa PO scraper and Payee Central site resolver

Email MBOX parser, Coupa web scraper, and Payee Central scrapers
for extracting PO data and resolving Amazon site codes.
This commit is contained in:
Adam Moussa 2026-04-29 19:44:46 -04:00
commit 17442e053a
10 changed files with 2744 additions and 0 deletions

6
.gitignore vendored Normal file
View file

@ -0,0 +1,6 @@
node_modules/
.env
output/
chrome-profile*/
cookies.json
.DS_Store

53
gen-po-list.mjs Normal file
View file

@ -0,0 +1,53 @@
import { DynamoDBClient } from "@aws-sdk/client-dynamodb";
import { DynamoDBDocumentClient, ScanCommand } from "@aws-sdk/lib-dynamodb";
import { readFileSync, writeFileSync, mkdirSync } from "fs";
import { execSync } from "child_process";
mkdirSync("output", { recursive: true });
// Get all PO numbers from DynamoDB
const client = new DynamoDBClient({ region: "us-east-1" });
const docClient = DynamoDBDocumentClient.from(client);
const dbPOs = new Set();
let lastKey = undefined;
console.log("Scanning DynamoDB for existing PO numbers...");
while (true) {
const resp = await docClient.send(
new ScanCommand({
TableName: "purchase-orders",
ProjectionExpression: "po_number",
ExclusiveStartKey: lastKey,
})
);
for (const item of resp.Items) {
dbPOs.add(item.po_number);
}
lastKey = resp.LastEvaluatedKey;
if (!lastKey) break;
}
console.log(`Found ${dbPOs.size} POs in DynamoDB`);
// Get POs from invoice spreadsheet using python helper
console.log("Extracting POs from invoice spreadsheet...");
const pyScript = [
"import openpyxl, json",
'wb = openpyxl.load_workbook("/Users/adammoussa/Documents/working-docs/plumbing-spend/invoices-2025.xlsx")',
"ws = wb.active",
"pos = set()",
"for row in ws.iter_rows(min_row=2, values_only=True):",
" if row[1]: pos.add(str(row[1]).strip())",
"print(json.dumps(sorted(list(pos))))",
].join("\n");
const invoicePOs = JSON.parse(
execSync(`python3.12 -c "${pyScript.replace(/"/g, '\\"')}"`, { encoding: "utf-8" })
);
console.log(`Found ${invoicePOs.length} unique POs in invoice file`);
// Find POs not in DB
const notInDb = invoicePOs.filter((po) => !dbPOs.has(po));
console.log(`POs not in DynamoDB: ${notInDb.length}`);
writeFileSync("output/po-list.json", JSON.stringify(notInDb, null, 2));
console.log("Saved to output/po-list.json");

163
monitor.mjs Normal file
View file

@ -0,0 +1,163 @@
import { readFileSync, watchFile, existsSync } from "fs";
const OUTPUT_FILE = "output/scraped-pos.json";
const PROGRESS_FILE = "output/progress.json";
const POLL_MS = 5000;
let lastCount = 0;
function classify(description) {
if (!description) return null;
const d = description.toLowerCase();
if (!d.includes("plumbing")) return null;
if (d.includes("plumbing pm") || d.includes("pm and corrective")) return "PM";
return "Reactive";
}
function extractSite(record) {
if (record.site_code) return record.site_code;
for (const li of record.line_items || []) {
if (!li.description) continue;
const m = li.description.match(/^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b/);
if (m) return m[1];
}
if (record.ship_to_raw) {
const m = record.ship_to_raw.match(/\(([A-Z0-9]{3,5})\)/);
if (m) return m[1];
}
return "Unknown";
}
function analyze() {
if (!existsSync(OUTPUT_FILE)) {
console.log("Waiting for scrape output...");
return;
}
let data;
try {
data = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
} catch {
return;
}
if (data.length === lastCount) return;
lastCount = data.length;
let progress = {};
if (existsSync(PROGRESS_FILE)) {
try {
progress = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
} catch {}
}
const sites = {};
let pmTotal = 0;
let reactiveTotal = 0;
let pmCount = 0;
let reactiveCount = 0;
let nonPlumbing = 0;
const plumbingPOs = [];
for (const record of data) {
let isPlumbing = false;
let category = null;
for (const li of record.line_items || []) {
const cat = classify(li.description);
if (cat) {
isPlumbing = true;
category = cat;
break;
}
}
if (!isPlumbing) {
nonPlumbing++;
continue;
}
const site = extractSite(record);
const amount = parseFloat(
(record.line_items[0]?.total || "0").replace(/,/g, "")
);
if (!sites[site]) sites[site] = { pm: 0, reactive: 0, pmCount: 0, reactiveCount: 0 };
if (category === "PM") {
sites[site].pm += amount;
sites[site].pmCount++;
pmTotal += amount;
pmCount++;
} else {
sites[site].reactive += amount;
sites[site].reactiveCount++;
reactiveTotal += amount;
reactiveCount++;
}
plumbingPOs.push({ po: record.po_number, site, category, amount });
}
// Clear screen and print dashboard
console.clear();
console.log("╔══════════════════════════════════════════════════════════════╗");
console.log("║ COUPA PO SCRAPER — LIVE PLUMBING MONITOR ║");
console.log("╚══════════════════════════════════════════════════════════════╝");
console.log();
console.log(` Scraped: ${data.length} / 4,192 POs (${Object.keys(progress).length} completed)`);
console.log(` Plumbing: ${pmCount + reactiveCount} POs | Non-plumbing: ${nonPlumbing}`);
console.log();
console.log(" ┌──────────────┬──────────────────┬──────────────────┬──────────────────┐");
console.log(" │ Category │ PO Count │ Spend │ Avg per PO │");
console.log(" ├──────────────┼──────────────────┼──────────────────┼──────────────────┤");
console.log(
` │ PM │ ${String(pmCount).padStart(16)} │ $${pmTotal.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${(pmCount ? pmTotal / pmCount : 0).toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
);
console.log(
` │ Reactive │ ${String(reactiveCount).padStart(16)} │ $${reactiveTotal.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${(reactiveCount ? reactiveTotal / reactiveCount : 0).toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
);
const total = pmTotal + reactiveTotal;
const totalCount = pmCount + reactiveCount;
console.log(" ├──────────────┼──────────────────┼──────────────────┼──────────────────┤");
console.log(
` │ TOTAL │ ${String(totalCount).padStart(16)} │ $${total.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${(totalCount ? total / totalCount : 0).toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
);
console.log(" └──────────────┴──────────────────┴──────────────────┴──────────────────┘");
// Top sites
const sortedSites = Object.entries(sites)
.map(([site, d]) => ({ site, total: d.pm + d.reactive, ...d }))
.sort((a, b) => b.total - a.total);
if (sortedSites.length > 0) {
console.log();
console.log(" Top 15 Sites:");
console.log(" ┌──────────┬──────────────────┬──────────────────┬──────────────────┐");
console.log(" │ Site │ PM Spend │ Reactive Spend │ Total │");
console.log(" ├──────────┼──────────────────┼──────────────────┼──────────────────┤");
for (const s of sortedSites.slice(0, 15)) {
console.log(
` │ ${s.site.padEnd(8)} │ $${s.pm.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${s.reactive.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${s.total.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
);
}
console.log(" └──────────┴──────────────────┴──────────────────┴──────────────────┘");
}
// Last 5 plumbing POs scraped
if (plumbingPOs.length > 0) {
console.log();
console.log(" Last 5 plumbing POs:");
for (const p of plumbingPOs.slice(-5)) {
console.log(` ${p.po} ${p.site.padEnd(8)} ${p.category.padEnd(10)} $${p.amount.toLocaleString()}`);
}
}
console.log();
console.log(` Last updated: ${new Date().toLocaleTimeString()}`);
console.log(" Press Ctrl+C to stop monitoring.");
}
// Run immediately, then poll
analyze();
setInterval(analyze, POLL_MS);

1492
package-lock.json generated Normal file

File diff suppressed because it is too large Load diff

18
package.json Normal file
View file

@ -0,0 +1,18 @@
{
"name": "coupa-po-scraper",
"version": "1.0.0",
"description": "",
"main": "index.js",
"scripts": {
"test": "echo \"Error: no test specified\" && exit 1"
},
"keywords": [],
"author": "",
"license": "ISC",
"type": "commonjs",
"dependencies": {
"@aws-sdk/client-dynamodb": "^3.1039.0",
"@aws-sdk/lib-dynamodb": "^3.1039.0",
"playwright": "^1.59.1"
}
}

219
parse-mbox.py Normal file
View file

@ -0,0 +1,219 @@
#!/usr/bin/env python3.12
"""Parse Coupa PO emails from an MBOX file and extract line items + ship-to data."""
import mailbox
import email
import re
import json
import sys
import os
def parse_amount(s):
"""Parse amount string, handling formats like '1.0 EACH x 3,000.00' or '10,000.00'."""
s = s.strip()
if ' x ' in s.lower():
s = s.split(' x ')[-1].strip()
elif ' X ' in s:
s = s.split(' X ')[-1].strip()
return float(s.replace(',', ''))
def parse_po_email(body):
"""Extract PO data from a Coupa 'issued' email plaintext body."""
lines = body.split('\n')
lines = [l.strip() for l in lines]
# PO number from body
po_match = re.search(r'Purchase Order #?((?:2D|B187|FK)-\d+)', body)
if not po_match:
return None
po_number = po_match.group(1)
# Line items: between "Items" line and the coupahost URL
items_start = None
items_end = None
for i, line in enumerate(lines):
if line == 'Items' and items_start is None:
items_start = i + 1
if items_start and 'supplier.coupahost.com/orders/' in line:
items_end = i
break
line_items = []
if items_start and items_end:
item_lines = [l for l in lines[items_start:items_end] if l]
i = 0
while i < len(item_lines):
desc = item_lines[i]
# Skip if this line looks like a number/currency
if re.match(r'^[\d,]', desc) or desc == 'USD':
i += 1
continue
amount = None
# Look ahead for amount
for j in range(i + 1, min(i + 4, len(item_lines))):
candidate = item_lines[j]
if re.match(r'^[\d,]', candidate):
try:
amount = parse_amount(candidate)
except ValueError:
pass
break
line_items.append({
'description': desc,
'amount': amount,
})
# Skip past amount + currency lines
i = j + 1 if amount is not None else i + 1
while i < len(item_lines) and item_lines[i] == 'USD':
i += 1
# Key-value pairs after "More Detail"
kv = {}
detail_start = None
for i, line in enumerate(lines):
if line == 'More Detail':
detail_start = i + 1
break
if detail_start:
kv_keys = ['PO ID', 'Department', 'Status', 'Last Opened', 'Order Date',
'Acknowledged At', 'Revision Date', 'Payment Term', 'Req #']
for i in range(detail_start, len(lines)):
for k in kv_keys:
if lines[i] == k and i + 1 < len(lines):
kv[k] = lines[i + 1]
# Ship-to address: second "Shipping" section
shipping_count = 0
ship_to_lines = []
capturing = False
skipping_blanks = False
for i, line in enumerate(lines):
if line == 'Shipping':
shipping_count += 1
if shipping_count == 2:
capturing = True
skipping_blanks = True
continue
elif capturing:
if skipping_blanks and line == '':
continue
skipping_blanks = False
if line in ('', 'Ship To Address') or line.startswith('---') or line.startswith('http'):
break
if line in kv_keys or line == 'Supplier' or line == 'More Detail':
break
ship_to_lines.append(line)
ship_to_raw = '\n'.join(ship_to_lines) if ship_to_lines else None
# Extract site code from ship-to
site_code = None
if ship_to_raw:
# (CODE) pattern — e.g. "Amazon.com Services LLC (XSF2)"
m = re.search(r'\(([A-Z0-9]{3,5})\)', ship_to_raw)
if m:
site_code = m.group(1)
# "LLC - CODE" — e.g. "Amazon.com Services LLC - WUT9"
if not site_code:
m = re.search(r'(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b', ship_to_raw)
if m:
site_code = m.group(1)
# "ATTN: ... Station CODE" or "ATTN: ... DS - CODE"
if not site_code:
m = re.search(r'ATTN:.*?(?:Station|DS)\s*[-–]?\s*([A-Z0-9]{3,5})\b', ship_to_raw)
if m:
site_code = m.group(1)
# "CODE - Amazon" at start of first line
if not site_code:
m = re.match(r'^([A-Z0-9]{3,5})\s*-\s*Amazon', ship_to_raw)
if m:
site_code = m.group(1)
if not site_code and line_items:
for li in line_items:
m = re.match(r'^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b', li['description'])
if m:
site_code = m.group(1)
break
return {
'po_number': po_number,
'site_code': site_code,
'status': kv.get('Status'),
'order_date': kv.get('Order Date'),
'revision_date': kv.get('Revision Date'),
'payment_term': kv.get('Payment Term'),
'req_number': kv.get('Req #'),
'ship_to_raw': ship_to_raw,
'line_items': line_items,
}
def process_mbox(mbox_path, target_pos=None, output_path=None):
"""Process an MBOX file and return parsed PO records."""
mbox = mailbox.mbox(mbox_path)
results = {}
processed = 0
matched = 0
for msg in mbox:
processed += 1
subject = msg.get('Subject', '')
if 'issued' not in subject.lower() or 'Purchase Order' not in subject:
continue
if msg.is_multipart():
body = None
for part in msg.walk():
if part.get_content_type() == 'text/plain':
body = part.get_payload(decode=True).decode('utf-8', errors='replace')
break
if not body:
continue
else:
body = msg.get_payload(decode=True).decode('utf-8', errors='replace')
record = parse_po_email(body)
if not record:
continue
po = record['po_number']
if target_pos and po not in target_pos:
continue
# Keep latest version if duplicate
if po not in results:
results[po] = record
matched += 1
if matched % 500 == 0 and matched > 0:
print(f" Matched {matched} POs so far... ({processed} messages scanned)", file=sys.stderr)
mbox.close()
result_list = list(results.values())
print(f"Processed {processed} messages, matched {matched} unique POs", file=sys.stderr)
if output_path:
with open(output_path, 'w') as f:
json.dump(result_list, f, indent=2)
print(f"Saved to {output_path}", file=sys.stderr)
return result_list
if __name__ == '__main__':
mbox_path = sys.argv[1] if len(sys.argv) > 1 else '/tmp/coupa-emails/coupa-po-dump--info@seahavenind.com-wHxmHD.mbox'
po_list_path = sys.argv[2] if len(sys.argv) > 2 else 'output/po-list.json'
output_path = sys.argv[3] if len(sys.argv) > 3 else 'output/email-parsed-pos.json'
target_pos = None
if os.path.exists(po_list_path):
with open(po_list_path) as f:
target_pos = set(json.load(f))
print(f"Targeting {len(target_pos)} specific POs", file=sys.stderr)
process_mbox(mbox_path, target_pos, output_path)

172
scrape-payee-chunk.mjs Normal file
View file

@ -0,0 +1,172 @@
import { chromium } from "playwright";
import { readFileSync, writeFileSync, existsSync } from "fs";
const chunkId = process.argv[2];
if (!chunkId && chunkId !== "0") {
console.error("Usage: node scrape-payee-chunk.mjs <chunk-id>");
process.exit(1);
}
const PO_LIST = `output/not-found-2026-chunk${chunkId}.json`;
const OUTPUT_FILE = `output/payee-2026-chunk${chunkId}-results.json`;
const PROGRESS_FILE = `output/payee-2026-chunk${chunkId}-progress.json`;
const PROFILE_DIR = `chrome-profile-${chunkId}`;
const DELAY_MS = 800;
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
console.log(`[Chunk ${chunkId}] Loaded ${allPOs.length} POs`);
let completed = {};
if (existsSync(PROGRESS_FILE)) {
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
console.log(`[Chunk ${chunkId}] Resuming: ${Object.keys(completed).length} already done`);
}
const remaining = allPOs.filter((po) => !completed[po]);
console.log(`[Chunk ${chunkId}] Remaining: ${remaining.length} POs\n`);
if (remaining.length === 0) {
console.log("All POs already scraped!");
process.exit(0);
}
function poUrl(po) {
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
}
// Copy chrome-profile if this chunk's profile doesn't exist
import { execSync } from "child_process";
const profilePath = new URL(`./${PROFILE_DIR}`, import.meta.url).pathname;
const mainProfile = new URL("./chrome-profile", import.meta.url).pathname;
if (!existsSync(profilePath)) {
console.log(`Copying chrome profile to ${PROFILE_DIR}...`);
execSync(`cp -R "${mainProfile}" "${profilePath}"`);
}
const context = await chromium.launchPersistentContext(profilePath, {
headless: false,
channel: "chrome",
args: ["--disable-blink-features=AutomationControlled"],
ignoreDefaultArgs: ["--enable-automation"],
});
const page = context.pages()[0] || (await context.newPage());
console.log(`[Chunk ${chunkId}] Opening Payee Central...`);
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log(`\n[Chunk ${chunkId}] 🔐 Login required. Please sign in...`);
await page.waitForURL(
(u) => {
const s = u.toString();
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
},
{ timeout: 180000 }
);
console.log(`[Chunk ${chunkId}] ✅ Logged in!\n`);
}
let results = [];
if (existsSync(OUTPUT_FILE)) {
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
}
let scraped = 0;
let errors = 0;
let consecutiveErrors = 0;
const startTime = Date.now();
for (const po of remaining) {
try {
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log(`\n[Chunk ${chunkId}] 🔐 Session expired. Please log in again...`);
await page.waitForURL(
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
{ timeout: 180000 }
);
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
}
await page.waitForTimeout(500);
const bodyText = await page.innerText("body");
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
let siteCode = null;
if (shipTo) {
for (const re of [
/\(([A-Z0-9]{3,5})\)/,
/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/,
/LLC\s+([A-Z0-9]{3,5})\b/,
/^([A-Z0-9]{3,5})\s*-\s*Amazon/m,
/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/,
]) {
const m = shipTo.match(re);
if (m) { siteCode = m[1]; break; }
}
}
const lineItems = await page.$$eval(
"table tbody tr",
(rows) =>
rows.map((row) => {
const cells = row.querySelectorAll("td");
if (cells.length < 7) return null;
return {
description: cells[1]?.textContent?.trim() || null,
need_by: cells[2]?.textContent?.trim() || null,
quantity: cells[3]?.textContent?.trim() || null,
unit: cells[4]?.textContent?.trim() || null,
price: cells[5]?.textContent?.trim() || null,
amount: cells[6]?.textContent?.trim() || null,
};
}).filter(Boolean)
).catch(() => []);
results.push({
po_number: po,
site_code: siteCode,
ship_to_raw: shipTo,
line_items: lineItems,
});
completed[po] = true;
scraped++;
consecutiveErrors = 0;
const elapsed = (Date.now() - startTime) / 1000;
const rate = scraped / elapsed;
const eta = Math.round((remaining.length - scraped) / rate);
const etaMin = Math.floor(eta / 60);
const etaSec = eta % 60;
process.stdout.write(
`\r[C${chunkId}] [${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s `
);
if (scraped % 10 === 0) {
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
}
} catch (err) {
console.error(`\n[C${chunkId}] [${po}] Error: ${err.message}`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
console.error("Too many consecutive errors, stopping.");
break;
}
}
await page.waitForTimeout(DELAY_MS);
}
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
console.log(`\n\n[Chunk ${chunkId}] Done! Scraped: ${scraped}, Errors: ${errors}`);
await context.close();

170
scrape-payee-notfound.mjs Normal file
View file

@ -0,0 +1,170 @@
import { chromium } from "playwright";
import { readFileSync, writeFileSync, existsSync } from "fs";
const PO_LIST = "output/not-found-pos.json";
const OUTPUT_FILE = "output/notfound-payee-results.json";
const PROGRESS_FILE = "output/notfound-payee-progress.json";
const DELAY_MS = 800;
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`);
let completed = {};
if (existsSync(PROGRESS_FILE)) {
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
}
const remaining = allPOs.filter((po) => !completed[po]);
console.log(`Remaining: ${remaining.length} POs\n`);
if (remaining.length === 0) {
console.log("All POs already scraped!");
process.exit(0);
}
function poUrl(po) {
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
}
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
const context = await chromium.launchPersistentContext(userDataDir, {
headless: false,
channel: "chrome",
args: ["--disable-blink-features=AutomationControlled"],
ignoreDefaultArgs: ["--enable-automation"],
});
const page = context.pages()[0] || (await context.newPage());
console.log("Opening Payee Central...");
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Login required. Please sign in in the browser window...");
await page.waitForURL(
(u) => {
const s = u.toString();
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
},
{ timeout: 180000 }
);
console.log("✅ Logged in!\n");
}
let results = [];
if (existsSync(OUTPUT_FILE)) {
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
}
let scraped = 0;
let errors = 0;
let consecutiveErrors = 0;
const startTime = Date.now();
for (const po of remaining) {
try {
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Session expired. Please log in again...");
await page.waitForURL(
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
{ timeout: 180000 }
);
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
}
await page.waitForTimeout(500);
const bodyText = await page.innerText("body");
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
let siteCode = null;
if (shipTo) {
const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/);
if (m) siteCode = m[1];
if (!siteCode) {
const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/);
if (m2) siteCode = m2[1];
}
if (!siteCode) {
const m3 = shipTo.match(/LLC\s+([A-Z0-9]{3,5})\b/);
if (m3) siteCode = m3[1];
}
if (!siteCode) {
const m4 = shipTo.match(/^([A-Z0-9]{3,5})\s*-\s*Amazon/m);
if (m4) siteCode = m4[1];
}
if (!siteCode) {
const m5 = shipTo.match(/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/);
if (m5) siteCode = m5[1];
}
}
const lineItems = await page.$$eval(
"table tbody tr",
(rows) =>
rows.map((row) => {
const cells = row.querySelectorAll("td");
if (cells.length < 7) return null;
return {
description: cells[1]?.textContent?.trim() || null,
need_by: cells[2]?.textContent?.trim() || null,
quantity: cells[3]?.textContent?.trim() || null,
unit: cells[4]?.textContent?.trim() || null,
price: cells[5]?.textContent?.trim() || null,
amount: cells[6]?.textContent?.trim() || null,
};
}).filter(Boolean)
).catch(() => []);
const result = {
po_number: po,
site_code: siteCode,
ship_to_raw: shipTo,
line_items: lineItems,
};
results.push(result);
completed[po] = true;
scraped++;
consecutiveErrors = 0;
const elapsed = (Date.now() - startTime) / 1000;
const rate = scraped / elapsed;
const eta = Math.round((remaining.length - scraped) / rate);
const etaMin = Math.floor(eta / 60);
const etaSec = eta % 60;
const desc0 = lineItems[0]?.description?.substring(0, 40) || "no items";
process.stdout.write(
`\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ${desc0} | ETA: ${etaMin}m${etaSec}s `
);
if (scraped % 10 === 0) {
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
}
} catch (err) {
console.error(`\n [${po}] Error: ${err.message}`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
console.error("Too many consecutive errors, stopping.");
break;
}
}
await page.waitForTimeout(DELAY_MS);
}
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
console.log(`Total in output: ${results.length} POs`);
console.log(`Results saved to ${OUTPUT_FILE}`);
await context.close();

159
scrape-payee.mjs Normal file
View file

@ -0,0 +1,159 @@
import { chromium } from "playwright";
import { readFileSync, writeFileSync, existsSync } from "fs";
const PO_LIST = "output/unknown-pos.json";
const OUTPUT_FILE = "output/payee-results.json";
const PROGRESS_FILE = "output/payee-progress.json";
const DELAY_MS = 800;
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`);
let completed = {};
if (existsSync(PROGRESS_FILE)) {
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
}
const remaining = allPOs.filter((po) => !completed[po]);
console.log(`Remaining: ${remaining.length} POs\n`);
if (remaining.length === 0) {
console.log("All POs already scraped!");
process.exit(0);
}
function poUrl(po) {
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
}
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
const context = await chromium.launchPersistentContext(userDataDir, {
headless: false,
channel: "chrome",
args: ["--disable-blink-features=AutomationControlled"],
ignoreDefaultArgs: ["--enable-automation"],
});
const page = context.pages()[0] || (await context.newPage());
console.log("Opening Payee Central...");
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Login required. Please sign in in the browser window...");
await page.waitForURL(
(u) => {
const s = u.toString();
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
},
{ timeout: 180000 }
);
console.log("✅ Logged in!\n");
}
let results = [];
if (existsSync(OUTPUT_FILE)) {
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
}
let scraped = 0;
let errors = 0;
let consecutiveErrors = 0;
const startTime = Date.now();
for (const po of remaining) {
try {
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log("\n🔐 Session expired. Please log in again...");
await page.waitForURL(
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
{ timeout: 180000 }
);
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
}
await page.waitForTimeout(500);
const bodyText = await page.innerText("body");
// Extract Ship To with site code
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
let siteCode = null;
if (shipTo) {
const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/);
if (m) siteCode = m[1];
if (!siteCode) {
const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/);
if (m2) siteCode = m2[1];
}
}
// Extract line items from table
const lineItems = await page.$$eval(
"table tbody tr",
(rows) =>
rows.map((row) => {
const cells = row.querySelectorAll("td");
if (cells.length < 7) return null;
return {
description: cells[1]?.textContent?.trim() || null,
need_by: cells[2]?.textContent?.trim() || null,
quantity: cells[3]?.textContent?.trim() || null,
unit: cells[4]?.textContent?.trim() || null,
price: cells[5]?.textContent?.trim() || null,
amount: cells[6]?.textContent?.trim() || null,
};
}).filter(Boolean)
).catch(() => []);
const result = {
po_number: po,
site_code: siteCode,
ship_to_raw: shipTo,
line_items: lineItems,
};
results.push(result);
completed[po] = true;
scraped++;
consecutiveErrors = 0;
const elapsed = (Date.now() - startTime) / 1000;
const rate = scraped / elapsed;
const eta = Math.round((remaining.length - scraped) / rate);
const etaMin = Math.floor(eta / 60);
const etaSec = eta % 60;
process.stdout.write(
`\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s `
);
if (scraped % 10 === 0) {
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
}
} catch (err) {
console.error(`\n [${po}] Error: ${err.message}`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
console.error("Too many consecutive errors, stopping.");
break;
}
}
await page.waitForTimeout(DELAY_MS);
}
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
console.log(`Total in output: ${results.length} POs`);
console.log(`Results saved to ${OUTPUT_FILE}`);
await context.close();

292
scrape.mjs Normal file
View file

@ -0,0 +1,292 @@
import { chromium } from "playwright";
import { readFileSync, writeFileSync, existsSync } from "fs";
const EMAIL = process.env.COUPA_EMAIL;
const PASSWORD = process.env.COUPA_PASSWORD;
if (!EMAIL || !PASSWORD) {
console.error("Set COUPA_EMAIL and COUPA_PASSWORD env vars");
process.exit(1);
}
const SUPPLIER_ID = "895025";
const INSTANCE_URL = "https%3A%2F%2Famazon.coupahost.com%2F";
const OUTPUT_FILE = "output/scraped-pos.json";
const PROGRESS_FILE = "output/progress.json";
const DELAY_MS = 1500;
const KEEPALIVE_EVERY = 75;
const poListFile = process.argv[2];
if (!poListFile) {
console.error("Usage: node scrape.mjs <po-list.json>");
process.exit(1);
}
const allPOs = JSON.parse(readFileSync(poListFile, "utf-8"));
console.log(`Loaded ${allPOs.length} POs to scrape`);
let completed = {};
if (existsSync(PROGRESS_FILE)) {
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
}
const remaining = allPOs.filter((po) => !completed[po]);
console.log(`Remaining: ${remaining.length} POs\n`);
if (remaining.length === 0) {
console.log("All POs already scraped!");
process.exit(0);
}
function poUrl(po) {
const numeric = po.split("-").slice(1).join("-");
return `https://supplier.coupahost.com/orders/${numeric}?instance_url=${INSTANCE_URL}&supplier_id=${SUPPLIER_ID}&`;
}
function isLoginPage(url) {
return url.includes("sessions/new") || url.includes("login") || url.includes("sign_in");
}
async function autoLogin(page) {
console.log("\n🔐 Session expired — auto-filling login form...");
// Step 1: Fill email and click Continue
try {
await page.waitForSelector("input#email", { state: "visible", timeout: 5000 });
await page.fill("input#email", EMAIL);
await page.click("button.s-login").catch(() => {});
console.log(" Email filled, clicked Continue.");
} catch {
// Already past email step
}
// Step 2: CAPTCHA appears before password — wait for user to solve it
// After CAPTCHA, password field will appear
console.log(" ⏳ Solve the CAPTCHA — password will auto-fill after...");
try {
await page.waitForSelector("input#password", { state: "visible", timeout: 120000 });
await page.fill("input#password", PASSWORD);
console.log(" Password filled, clicking Login...");
page.click("button.s-login").catch(() => {});
} catch {
console.log(" Password field never appeared. Check the browser.");
}
// Wait for login to complete (may need another CAPTCHA or just redirects)
console.log(" ⏳ Waiting for login to complete...");
try {
await page.waitForURL((url) => !isLoginPage(url.toString()), { timeout: 120000 });
console.log(" ✅ Login successful!\n");
return true;
} catch {
console.error(" ❌ Login timed out after 120s. Exiting.");
return false;
}
}
async function keepAlive(page) {
try {
await page.goto("https://supplier.coupahost.com/home/", {
waitUntil: "domcontentloaded",
timeout: 15000,
});
if (isLoginPage(page.url())) return false;
return true;
} catch {
return false;
}
}
async function scrapePO(page, po) {
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 30000 });
if (isLoginPage(page.url())) {
const ok = await autoLogin(page);
if (!ok) return null;
// Navigate to PO after re-login
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 30000 });
if (isLoginPage(page.url())) return null;
}
await page.waitForSelector("iframe#enterprise_frame", { timeout: 15000 });
await page.waitForTimeout(1500);
const frame =
page.frame({ name: "enterprise_frame" }) ||
page.frames().find((f) => f.url().includes("supplier_order_headers"));
if (!frame) return "no_iframe";
await frame.waitForLoadState("networkidle");
const data = await frame.evaluate(() => {
const text = (sel) => {
const el = document.querySelector(sel);
return el ? el.textContent.trim() : null;
};
const lineItems = [];
const rows = document.querySelectorAll(".coupa_datatable_row");
for (const row of rows) {
const desc = row.querySelector(".s-description");
const price = row.querySelector(".s-price span[title]");
const total = row.querySelector(".s-total span[title]");
const needBy = row.querySelector("[id$='_need_by_date']");
lineItems.push({
description: desc ? desc.textContent.trim() : null,
price: price ? price.textContent.trim() : null,
total: total ? total.textContent.trim() : null,
need_by: needBy ? needBy.textContent.trim() : null,
});
}
const addrEl = document.querySelector(
"label[for='order_header_ship_to_address'] + span.address, span.address"
);
let shipTo = null;
if (addrEl) {
const clone = addrEl.cloneNode(true);
clone.querySelectorAll(".form_element").forEach((el) => el.remove());
shipTo = clone.innerHTML
.replace(/<br\s*\/?>/gi, "\n")
.replace(/<[^>]+>/g, "")
.trim();
}
return {
status: text(".s-readable_status"),
order_date: text(".s-local_header_created_at"),
requester: text(".s-requester"),
payment_term: text("#order_header_payment_term .data"),
ship_to_raw: shipTo,
line_items: lineItems,
};
});
let site_code = null;
if (data.ship_to_raw) {
const m = data.ship_to_raw.match(/\(([A-Z0-9]{3,5})\)/);
if (m) site_code = m[1];
if (!site_code) {
const m2 = data.ship_to_raw.match(/Attn:\s*([A-Z0-9]{3,5})\b/);
if (m2) site_code = m2[1];
}
}
return { po_number: po, site_code, ...data };
}
// --- Main ---
// Use a persistent Chrome profile so reCAPTCHA trusts the browser
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
const context = await chromium.launchPersistentContext(userDataDir, {
headless: false,
channel: "chrome",
args: ["--disable-blink-features=AutomationControlled"],
ignoreDefaultArgs: ["--enable-automation"],
});
const page = context.pages()[0] || (await context.newPage());
// Initial login
console.log("Opening Coupa supplier portal...");
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
if (isLoginPage(page.url())) {
const ok = await autoLogin(page);
if (!ok) {
await context.close();
process.exit(1);
}
}
let results = [];
if (existsSync(OUTPUT_FILE)) {
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
}
let scraped = 0;
let errors = 0;
let consecutiveErrors = 0;
const startTime = Date.now();
for (const po of remaining) {
// Session keep-alive
if (scraped > 0 && scraped % KEEPALIVE_EVERY === 0) {
process.stdout.write("\n Refreshing session...");
const alive = await keepAlive(page);
if (!alive) {
console.log(" session expired, re-logging in...");
await page.goto("https://supplier.coupahost.com/sessions/new", {
waitUntil: "networkidle",
});
const ok = await autoLogin(page);
if (!ok) break;
} else {
process.stdout.write(" ok\n");
}
}
try {
const result = await scrapePO(page, po);
if (result === null) {
console.error(`\nFailed to authenticate. Saving progress.`);
break;
}
if (result === "no_iframe") {
console.error(`\n [${po}] No iframe found, skipping`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
console.error("Too many consecutive errors, stopping.");
break;
}
continue;
}
results.push(result);
completed[po] = true;
scraped++;
consecutiveErrors = 0;
const elapsed = (Date.now() - startTime) / 1000;
const rate = scraped / elapsed;
const eta = Math.round((remaining.length - scraped) / rate);
const etaMin = Math.floor(eta / 60);
const etaSec = eta % 60;
const desc0 = result.line_items[0]?.description?.substring(0, 50) || "no lines";
process.stdout.write(
`\r[${scraped}/${remaining.length}] ${po} | ${result.site_code || "?"} | ${desc0}... | ETA: ${etaMin}m${etaSec}s `
);
// Save every 10 POs (more frequent for monitor)
if (scraped % 10 === 0) {
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
}
} catch (err) {
console.error(`\n [${po}] Error: ${err.message}`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
console.error("Too many consecutive errors, stopping.");
break;
}
}
await page.waitForTimeout(DELAY_MS);
}
// Final save
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
console.log(`Total in output: ${results.length} POs`);
console.log(`Results saved to ${OUTPUT_FILE}`);
await context.close();