Initial commit: Coupa PO scraper and Payee Central site resolver
Email MBOX parser, Coupa web scraper, and Payee Central scrapers for extracting PO data and resolving Amazon site codes.
This commit is contained in:
commit
17442e053a
10 changed files with 2744 additions and 0 deletions
6
.gitignore
vendored
Normal file
6
.gitignore
vendored
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
node_modules/
|
||||
.env
|
||||
output/
|
||||
chrome-profile*/
|
||||
cookies.json
|
||||
.DS_Store
|
||||
53
gen-po-list.mjs
Normal file
53
gen-po-list.mjs
Normal file
|
|
@ -0,0 +1,53 @@
|
|||
import { DynamoDBClient } from "@aws-sdk/client-dynamodb";
|
||||
import { DynamoDBDocumentClient, ScanCommand } from "@aws-sdk/lib-dynamodb";
|
||||
import { readFileSync, writeFileSync, mkdirSync } from "fs";
|
||||
import { execSync } from "child_process";
|
||||
|
||||
mkdirSync("output", { recursive: true });
|
||||
|
||||
// Get all PO numbers from DynamoDB
|
||||
const client = new DynamoDBClient({ region: "us-east-1" });
|
||||
const docClient = DynamoDBDocumentClient.from(client);
|
||||
|
||||
const dbPOs = new Set();
|
||||
let lastKey = undefined;
|
||||
|
||||
console.log("Scanning DynamoDB for existing PO numbers...");
|
||||
while (true) {
|
||||
const resp = await docClient.send(
|
||||
new ScanCommand({
|
||||
TableName: "purchase-orders",
|
||||
ProjectionExpression: "po_number",
|
||||
ExclusiveStartKey: lastKey,
|
||||
})
|
||||
);
|
||||
for (const item of resp.Items) {
|
||||
dbPOs.add(item.po_number);
|
||||
}
|
||||
lastKey = resp.LastEvaluatedKey;
|
||||
if (!lastKey) break;
|
||||
}
|
||||
console.log(`Found ${dbPOs.size} POs in DynamoDB`);
|
||||
|
||||
// Get POs from invoice spreadsheet using python helper
|
||||
console.log("Extracting POs from invoice spreadsheet...");
|
||||
const pyScript = [
|
||||
"import openpyxl, json",
|
||||
'wb = openpyxl.load_workbook("/Users/adammoussa/Documents/working-docs/plumbing-spend/invoices-2025.xlsx")',
|
||||
"ws = wb.active",
|
||||
"pos = set()",
|
||||
"for row in ws.iter_rows(min_row=2, values_only=True):",
|
||||
" if row[1]: pos.add(str(row[1]).strip())",
|
||||
"print(json.dumps(sorted(list(pos))))",
|
||||
].join("\n");
|
||||
const invoicePOs = JSON.parse(
|
||||
execSync(`python3.12 -c "${pyScript.replace(/"/g, '\\"')}"`, { encoding: "utf-8" })
|
||||
);
|
||||
console.log(`Found ${invoicePOs.length} unique POs in invoice file`);
|
||||
|
||||
// Find POs not in DB
|
||||
const notInDb = invoicePOs.filter((po) => !dbPOs.has(po));
|
||||
console.log(`POs not in DynamoDB: ${notInDb.length}`);
|
||||
|
||||
writeFileSync("output/po-list.json", JSON.stringify(notInDb, null, 2));
|
||||
console.log("Saved to output/po-list.json");
|
||||
163
monitor.mjs
Normal file
163
monitor.mjs
Normal file
|
|
@ -0,0 +1,163 @@
|
|||
import { readFileSync, watchFile, existsSync } from "fs";
|
||||
|
||||
const OUTPUT_FILE = "output/scraped-pos.json";
|
||||
const PROGRESS_FILE = "output/progress.json";
|
||||
const POLL_MS = 5000;
|
||||
|
||||
let lastCount = 0;
|
||||
|
||||
function classify(description) {
|
||||
if (!description) return null;
|
||||
const d = description.toLowerCase();
|
||||
if (!d.includes("plumbing")) return null;
|
||||
if (d.includes("plumbing pm") || d.includes("pm and corrective")) return "PM";
|
||||
return "Reactive";
|
||||
}
|
||||
|
||||
function extractSite(record) {
|
||||
if (record.site_code) return record.site_code;
|
||||
for (const li of record.line_items || []) {
|
||||
if (!li.description) continue;
|
||||
const m = li.description.match(/^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b/);
|
||||
if (m) return m[1];
|
||||
}
|
||||
if (record.ship_to_raw) {
|
||||
const m = record.ship_to_raw.match(/\(([A-Z0-9]{3,5})\)/);
|
||||
if (m) return m[1];
|
||||
}
|
||||
return "Unknown";
|
||||
}
|
||||
|
||||
function analyze() {
|
||||
if (!existsSync(OUTPUT_FILE)) {
|
||||
console.log("Waiting for scrape output...");
|
||||
return;
|
||||
}
|
||||
|
||||
let data;
|
||||
try {
|
||||
data = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
|
||||
if (data.length === lastCount) return;
|
||||
lastCount = data.length;
|
||||
|
||||
let progress = {};
|
||||
if (existsSync(PROGRESS_FILE)) {
|
||||
try {
|
||||
progress = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
|
||||
} catch {}
|
||||
}
|
||||
|
||||
const sites = {};
|
||||
let pmTotal = 0;
|
||||
let reactiveTotal = 0;
|
||||
let pmCount = 0;
|
||||
let reactiveCount = 0;
|
||||
let nonPlumbing = 0;
|
||||
const plumbingPOs = [];
|
||||
|
||||
for (const record of data) {
|
||||
let isPlumbing = false;
|
||||
let category = null;
|
||||
|
||||
for (const li of record.line_items || []) {
|
||||
const cat = classify(li.description);
|
||||
if (cat) {
|
||||
isPlumbing = true;
|
||||
category = cat;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!isPlumbing) {
|
||||
nonPlumbing++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const site = extractSite(record);
|
||||
const amount = parseFloat(
|
||||
(record.line_items[0]?.total || "0").replace(/,/g, "")
|
||||
);
|
||||
|
||||
if (!sites[site]) sites[site] = { pm: 0, reactive: 0, pmCount: 0, reactiveCount: 0 };
|
||||
|
||||
if (category === "PM") {
|
||||
sites[site].pm += amount;
|
||||
sites[site].pmCount++;
|
||||
pmTotal += amount;
|
||||
pmCount++;
|
||||
} else {
|
||||
sites[site].reactive += amount;
|
||||
sites[site].reactiveCount++;
|
||||
reactiveTotal += amount;
|
||||
reactiveCount++;
|
||||
}
|
||||
|
||||
plumbingPOs.push({ po: record.po_number, site, category, amount });
|
||||
}
|
||||
|
||||
// Clear screen and print dashboard
|
||||
console.clear();
|
||||
console.log("╔══════════════════════════════════════════════════════════════╗");
|
||||
console.log("║ COUPA PO SCRAPER — LIVE PLUMBING MONITOR ║");
|
||||
console.log("╚══════════════════════════════════════════════════════════════╝");
|
||||
console.log();
|
||||
console.log(` Scraped: ${data.length} / 4,192 POs (${Object.keys(progress).length} completed)`);
|
||||
console.log(` Plumbing: ${pmCount + reactiveCount} POs | Non-plumbing: ${nonPlumbing}`);
|
||||
console.log();
|
||||
console.log(" ┌──────────────┬──────────────────┬──────────────────┬──────────────────┐");
|
||||
console.log(" │ Category │ PO Count │ Spend │ Avg per PO │");
|
||||
console.log(" ├──────────────┼──────────────────┼──────────────────┼──────────────────┤");
|
||||
console.log(
|
||||
` │ PM │ ${String(pmCount).padStart(16)} │ $${pmTotal.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${(pmCount ? pmTotal / pmCount : 0).toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
|
||||
);
|
||||
console.log(
|
||||
` │ Reactive │ ${String(reactiveCount).padStart(16)} │ $${reactiveTotal.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${(reactiveCount ? reactiveTotal / reactiveCount : 0).toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
|
||||
);
|
||||
const total = pmTotal + reactiveTotal;
|
||||
const totalCount = pmCount + reactiveCount;
|
||||
console.log(" ├──────────────┼──────────────────┼──────────────────┼──────────────────┤");
|
||||
console.log(
|
||||
` │ TOTAL │ ${String(totalCount).padStart(16)} │ $${total.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${(totalCount ? total / totalCount : 0).toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
|
||||
);
|
||||
console.log(" └──────────────┴──────────────────┴──────────────────┴──────────────────┘");
|
||||
|
||||
// Top sites
|
||||
const sortedSites = Object.entries(sites)
|
||||
.map(([site, d]) => ({ site, total: d.pm + d.reactive, ...d }))
|
||||
.sort((a, b) => b.total - a.total);
|
||||
|
||||
if (sortedSites.length > 0) {
|
||||
console.log();
|
||||
console.log(" Top 15 Sites:");
|
||||
console.log(" ┌──────────┬──────────────────┬──────────────────┬──────────────────┐");
|
||||
console.log(" │ Site │ PM Spend │ Reactive Spend │ Total │");
|
||||
console.log(" ├──────────┼──────────────────┼──────────────────┼──────────────────┤");
|
||||
for (const s of sortedSites.slice(0, 15)) {
|
||||
console.log(
|
||||
` │ ${s.site.padEnd(8)} │ $${s.pm.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${s.reactive.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │ $${s.total.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 }).padStart(15)} │`
|
||||
);
|
||||
}
|
||||
console.log(" └──────────┴──────────────────┴──────────────────┴──────────────────┘");
|
||||
}
|
||||
|
||||
// Last 5 plumbing POs scraped
|
||||
if (plumbingPOs.length > 0) {
|
||||
console.log();
|
||||
console.log(" Last 5 plumbing POs:");
|
||||
for (const p of plumbingPOs.slice(-5)) {
|
||||
console.log(` ${p.po} ${p.site.padEnd(8)} ${p.category.padEnd(10)} $${p.amount.toLocaleString()}`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log();
|
||||
console.log(` Last updated: ${new Date().toLocaleTimeString()}`);
|
||||
console.log(" Press Ctrl+C to stop monitoring.");
|
||||
}
|
||||
|
||||
// Run immediately, then poll
|
||||
analyze();
|
||||
setInterval(analyze, POLL_MS);
|
||||
1492
package-lock.json
generated
Normal file
1492
package-lock.json
generated
Normal file
File diff suppressed because it is too large
Load diff
18
package.json
Normal file
18
package.json
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
{
|
||||
"name": "coupa-po-scraper",
|
||||
"version": "1.0.0",
|
||||
"description": "",
|
||||
"main": "index.js",
|
||||
"scripts": {
|
||||
"test": "echo \"Error: no test specified\" && exit 1"
|
||||
},
|
||||
"keywords": [],
|
||||
"author": "",
|
||||
"license": "ISC",
|
||||
"type": "commonjs",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-dynamodb": "^3.1039.0",
|
||||
"@aws-sdk/lib-dynamodb": "^3.1039.0",
|
||||
"playwright": "^1.59.1"
|
||||
}
|
||||
}
|
||||
219
parse-mbox.py
Normal file
219
parse-mbox.py
Normal file
|
|
@ -0,0 +1,219 @@
|
|||
#!/usr/bin/env python3.12
|
||||
"""Parse Coupa PO emails from an MBOX file and extract line items + ship-to data."""
|
||||
|
||||
import mailbox
|
||||
import email
|
||||
import re
|
||||
import json
|
||||
import sys
|
||||
import os
|
||||
|
||||
def parse_amount(s):
|
||||
"""Parse amount string, handling formats like '1.0 EACH x 3,000.00' or '10,000.00'."""
|
||||
s = s.strip()
|
||||
if ' x ' in s.lower():
|
||||
s = s.split(' x ')[-1].strip()
|
||||
elif ' X ' in s:
|
||||
s = s.split(' X ')[-1].strip()
|
||||
return float(s.replace(',', ''))
|
||||
|
||||
|
||||
def parse_po_email(body):
|
||||
"""Extract PO data from a Coupa 'issued' email plaintext body."""
|
||||
lines = body.split('\n')
|
||||
lines = [l.strip() for l in lines]
|
||||
|
||||
# PO number from body
|
||||
po_match = re.search(r'Purchase Order #?((?:2D|B187|FK)-\d+)', body)
|
||||
if not po_match:
|
||||
return None
|
||||
po_number = po_match.group(1)
|
||||
|
||||
# Line items: between "Items" line and the coupahost URL
|
||||
items_start = None
|
||||
items_end = None
|
||||
for i, line in enumerate(lines):
|
||||
if line == 'Items' and items_start is None:
|
||||
items_start = i + 1
|
||||
if items_start and 'supplier.coupahost.com/orders/' in line:
|
||||
items_end = i
|
||||
break
|
||||
|
||||
line_items = []
|
||||
if items_start and items_end:
|
||||
item_lines = [l for l in lines[items_start:items_end] if l]
|
||||
i = 0
|
||||
while i < len(item_lines):
|
||||
desc = item_lines[i]
|
||||
# Skip if this line looks like a number/currency
|
||||
if re.match(r'^[\d,]', desc) or desc == 'USD':
|
||||
i += 1
|
||||
continue
|
||||
amount = None
|
||||
# Look ahead for amount
|
||||
for j in range(i + 1, min(i + 4, len(item_lines))):
|
||||
candidate = item_lines[j]
|
||||
if re.match(r'^[\d,]', candidate):
|
||||
try:
|
||||
amount = parse_amount(candidate)
|
||||
except ValueError:
|
||||
pass
|
||||
break
|
||||
line_items.append({
|
||||
'description': desc,
|
||||
'amount': amount,
|
||||
})
|
||||
# Skip past amount + currency lines
|
||||
i = j + 1 if amount is not None else i + 1
|
||||
while i < len(item_lines) and item_lines[i] == 'USD':
|
||||
i += 1
|
||||
|
||||
# Key-value pairs after "More Detail"
|
||||
kv = {}
|
||||
detail_start = None
|
||||
for i, line in enumerate(lines):
|
||||
if line == 'More Detail':
|
||||
detail_start = i + 1
|
||||
break
|
||||
|
||||
if detail_start:
|
||||
kv_keys = ['PO ID', 'Department', 'Status', 'Last Opened', 'Order Date',
|
||||
'Acknowledged At', 'Revision Date', 'Payment Term', 'Req #']
|
||||
for i in range(detail_start, len(lines)):
|
||||
for k in kv_keys:
|
||||
if lines[i] == k and i + 1 < len(lines):
|
||||
kv[k] = lines[i + 1]
|
||||
|
||||
# Ship-to address: second "Shipping" section
|
||||
shipping_count = 0
|
||||
ship_to_lines = []
|
||||
capturing = False
|
||||
skipping_blanks = False
|
||||
for i, line in enumerate(lines):
|
||||
if line == 'Shipping':
|
||||
shipping_count += 1
|
||||
if shipping_count == 2:
|
||||
capturing = True
|
||||
skipping_blanks = True
|
||||
continue
|
||||
elif capturing:
|
||||
if skipping_blanks and line == '':
|
||||
continue
|
||||
skipping_blanks = False
|
||||
if line in ('', 'Ship To Address') or line.startswith('---') or line.startswith('http'):
|
||||
break
|
||||
if line in kv_keys or line == 'Supplier' or line == 'More Detail':
|
||||
break
|
||||
ship_to_lines.append(line)
|
||||
|
||||
ship_to_raw = '\n'.join(ship_to_lines) if ship_to_lines else None
|
||||
|
||||
# Extract site code from ship-to
|
||||
site_code = None
|
||||
if ship_to_raw:
|
||||
# (CODE) pattern — e.g. "Amazon.com Services LLC (XSF2)"
|
||||
m = re.search(r'\(([A-Z0-9]{3,5})\)', ship_to_raw)
|
||||
if m:
|
||||
site_code = m.group(1)
|
||||
# "LLC - CODE" — e.g. "Amazon.com Services LLC - WUT9"
|
||||
if not site_code:
|
||||
m = re.search(r'(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b', ship_to_raw)
|
||||
if m:
|
||||
site_code = m.group(1)
|
||||
# "ATTN: ... Station CODE" or "ATTN: ... DS - CODE"
|
||||
if not site_code:
|
||||
m = re.search(r'ATTN:.*?(?:Station|DS)\s*[-–]?\s*([A-Z0-9]{3,5})\b', ship_to_raw)
|
||||
if m:
|
||||
site_code = m.group(1)
|
||||
# "CODE - Amazon" at start of first line
|
||||
if not site_code:
|
||||
m = re.match(r'^([A-Z0-9]{3,5})\s*-\s*Amazon', ship_to_raw)
|
||||
if m:
|
||||
site_code = m.group(1)
|
||||
if not site_code and line_items:
|
||||
for li in line_items:
|
||||
m = re.match(r'^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b', li['description'])
|
||||
if m:
|
||||
site_code = m.group(1)
|
||||
break
|
||||
|
||||
return {
|
||||
'po_number': po_number,
|
||||
'site_code': site_code,
|
||||
'status': kv.get('Status'),
|
||||
'order_date': kv.get('Order Date'),
|
||||
'revision_date': kv.get('Revision Date'),
|
||||
'payment_term': kv.get('Payment Term'),
|
||||
'req_number': kv.get('Req #'),
|
||||
'ship_to_raw': ship_to_raw,
|
||||
'line_items': line_items,
|
||||
}
|
||||
|
||||
|
||||
def process_mbox(mbox_path, target_pos=None, output_path=None):
|
||||
"""Process an MBOX file and return parsed PO records."""
|
||||
mbox = mailbox.mbox(mbox_path)
|
||||
results = {}
|
||||
processed = 0
|
||||
matched = 0
|
||||
|
||||
for msg in mbox:
|
||||
processed += 1
|
||||
subject = msg.get('Subject', '')
|
||||
|
||||
if 'issued' not in subject.lower() or 'Purchase Order' not in subject:
|
||||
continue
|
||||
|
||||
if msg.is_multipart():
|
||||
body = None
|
||||
for part in msg.walk():
|
||||
if part.get_content_type() == 'text/plain':
|
||||
body = part.get_payload(decode=True).decode('utf-8', errors='replace')
|
||||
break
|
||||
if not body:
|
||||
continue
|
||||
else:
|
||||
body = msg.get_payload(decode=True).decode('utf-8', errors='replace')
|
||||
|
||||
record = parse_po_email(body)
|
||||
if not record:
|
||||
continue
|
||||
|
||||
po = record['po_number']
|
||||
|
||||
if target_pos and po not in target_pos:
|
||||
continue
|
||||
|
||||
# Keep latest version if duplicate
|
||||
if po not in results:
|
||||
results[po] = record
|
||||
matched += 1
|
||||
|
||||
if matched % 500 == 0 and matched > 0:
|
||||
print(f" Matched {matched} POs so far... ({processed} messages scanned)", file=sys.stderr)
|
||||
|
||||
mbox.close()
|
||||
|
||||
result_list = list(results.values())
|
||||
print(f"Processed {processed} messages, matched {matched} unique POs", file=sys.stderr)
|
||||
|
||||
if output_path:
|
||||
with open(output_path, 'w') as f:
|
||||
json.dump(result_list, f, indent=2)
|
||||
print(f"Saved to {output_path}", file=sys.stderr)
|
||||
|
||||
return result_list
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
mbox_path = sys.argv[1] if len(sys.argv) > 1 else '/tmp/coupa-emails/coupa-po-dump--info@seahavenind.com-wHxmHD.mbox'
|
||||
po_list_path = sys.argv[2] if len(sys.argv) > 2 else 'output/po-list.json'
|
||||
output_path = sys.argv[3] if len(sys.argv) > 3 else 'output/email-parsed-pos.json'
|
||||
|
||||
target_pos = None
|
||||
if os.path.exists(po_list_path):
|
||||
with open(po_list_path) as f:
|
||||
target_pos = set(json.load(f))
|
||||
print(f"Targeting {len(target_pos)} specific POs", file=sys.stderr)
|
||||
|
||||
process_mbox(mbox_path, target_pos, output_path)
|
||||
172
scrape-payee-chunk.mjs
Normal file
172
scrape-payee-chunk.mjs
Normal file
|
|
@ -0,0 +1,172 @@
|
|||
import { chromium } from "playwright";
|
||||
import { readFileSync, writeFileSync, existsSync } from "fs";
|
||||
|
||||
const chunkId = process.argv[2];
|
||||
if (!chunkId && chunkId !== "0") {
|
||||
console.error("Usage: node scrape-payee-chunk.mjs <chunk-id>");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const PO_LIST = `output/not-found-2026-chunk${chunkId}.json`;
|
||||
const OUTPUT_FILE = `output/payee-2026-chunk${chunkId}-results.json`;
|
||||
const PROGRESS_FILE = `output/payee-2026-chunk${chunkId}-progress.json`;
|
||||
const PROFILE_DIR = `chrome-profile-${chunkId}`;
|
||||
const DELAY_MS = 800;
|
||||
|
||||
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
|
||||
console.log(`[Chunk ${chunkId}] Loaded ${allPOs.length} POs`);
|
||||
|
||||
let completed = {};
|
||||
if (existsSync(PROGRESS_FILE)) {
|
||||
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
|
||||
console.log(`[Chunk ${chunkId}] Resuming: ${Object.keys(completed).length} already done`);
|
||||
}
|
||||
|
||||
const remaining = allPOs.filter((po) => !completed[po]);
|
||||
console.log(`[Chunk ${chunkId}] Remaining: ${remaining.length} POs\n`);
|
||||
|
||||
if (remaining.length === 0) {
|
||||
console.log("All POs already scraped!");
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
function poUrl(po) {
|
||||
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
|
||||
}
|
||||
|
||||
// Copy chrome-profile if this chunk's profile doesn't exist
|
||||
import { execSync } from "child_process";
|
||||
const profilePath = new URL(`./${PROFILE_DIR}`, import.meta.url).pathname;
|
||||
const mainProfile = new URL("./chrome-profile", import.meta.url).pathname;
|
||||
if (!existsSync(profilePath)) {
|
||||
console.log(`Copying chrome profile to ${PROFILE_DIR}...`);
|
||||
execSync(`cp -R "${mainProfile}" "${profilePath}"`);
|
||||
}
|
||||
|
||||
const context = await chromium.launchPersistentContext(profilePath, {
|
||||
headless: false,
|
||||
channel: "chrome",
|
||||
args: ["--disable-blink-features=AutomationControlled"],
|
||||
ignoreDefaultArgs: ["--enable-automation"],
|
||||
});
|
||||
|
||||
const page = context.pages()[0] || (await context.newPage());
|
||||
|
||||
console.log(`[Chunk ${chunkId}] Opening Payee Central...`);
|
||||
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
|
||||
|
||||
if (page.url().includes("signin") || page.url().includes("/ap/")) {
|
||||
console.log(`\n[Chunk ${chunkId}] 🔐 Login required. Please sign in...`);
|
||||
await page.waitForURL(
|
||||
(u) => {
|
||||
const s = u.toString();
|
||||
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
|
||||
},
|
||||
{ timeout: 180000 }
|
||||
);
|
||||
console.log(`[Chunk ${chunkId}] ✅ Logged in!\n`);
|
||||
}
|
||||
|
||||
let results = [];
|
||||
if (existsSync(OUTPUT_FILE)) {
|
||||
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
|
||||
}
|
||||
|
||||
let scraped = 0;
|
||||
let errors = 0;
|
||||
let consecutiveErrors = 0;
|
||||
const startTime = Date.now();
|
||||
|
||||
for (const po of remaining) {
|
||||
try {
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
|
||||
|
||||
if (page.url().includes("signin") || page.url().includes("/ap/")) {
|
||||
console.log(`\n[Chunk ${chunkId}] 🔐 Session expired. Please log in again...`);
|
||||
await page.waitForURL(
|
||||
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
|
||||
{ timeout: 180000 }
|
||||
);
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
|
||||
}
|
||||
|
||||
await page.waitForTimeout(500);
|
||||
const bodyText = await page.innerText("body");
|
||||
|
||||
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
|
||||
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
|
||||
|
||||
let siteCode = null;
|
||||
if (shipTo) {
|
||||
for (const re of [
|
||||
/\(([A-Z0-9]{3,5})\)/,
|
||||
/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/,
|
||||
/LLC\s+([A-Z0-9]{3,5})\b/,
|
||||
/^([A-Z0-9]{3,5})\s*-\s*Amazon/m,
|
||||
/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/,
|
||||
]) {
|
||||
const m = shipTo.match(re);
|
||||
if (m) { siteCode = m[1]; break; }
|
||||
}
|
||||
}
|
||||
|
||||
const lineItems = await page.$$eval(
|
||||
"table tbody tr",
|
||||
(rows) =>
|
||||
rows.map((row) => {
|
||||
const cells = row.querySelectorAll("td");
|
||||
if (cells.length < 7) return null;
|
||||
return {
|
||||
description: cells[1]?.textContent?.trim() || null,
|
||||
need_by: cells[2]?.textContent?.trim() || null,
|
||||
quantity: cells[3]?.textContent?.trim() || null,
|
||||
unit: cells[4]?.textContent?.trim() || null,
|
||||
price: cells[5]?.textContent?.trim() || null,
|
||||
amount: cells[6]?.textContent?.trim() || null,
|
||||
};
|
||||
}).filter(Boolean)
|
||||
).catch(() => []);
|
||||
|
||||
results.push({
|
||||
po_number: po,
|
||||
site_code: siteCode,
|
||||
ship_to_raw: shipTo,
|
||||
line_items: lineItems,
|
||||
});
|
||||
|
||||
completed[po] = true;
|
||||
scraped++;
|
||||
consecutiveErrors = 0;
|
||||
|
||||
const elapsed = (Date.now() - startTime) / 1000;
|
||||
const rate = scraped / elapsed;
|
||||
const eta = Math.round((remaining.length - scraped) / rate);
|
||||
const etaMin = Math.floor(eta / 60);
|
||||
const etaSec = eta % 60;
|
||||
|
||||
process.stdout.write(
|
||||
`\r[C${chunkId}] [${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s `
|
||||
);
|
||||
|
||||
if (scraped % 10 === 0) {
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`\n[C${chunkId}] [${po}] Error: ${err.message}`);
|
||||
errors++;
|
||||
consecutiveErrors++;
|
||||
if (consecutiveErrors > 10) {
|
||||
console.error("Too many consecutive errors, stopping.");
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
await page.waitForTimeout(DELAY_MS);
|
||||
}
|
||||
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
|
||||
console.log(`\n\n[Chunk ${chunkId}] Done! Scraped: ${scraped}, Errors: ${errors}`);
|
||||
await context.close();
|
||||
170
scrape-payee-notfound.mjs
Normal file
170
scrape-payee-notfound.mjs
Normal file
|
|
@ -0,0 +1,170 @@
|
|||
import { chromium } from "playwright";
|
||||
import { readFileSync, writeFileSync, existsSync } from "fs";
|
||||
|
||||
const PO_LIST = "output/not-found-pos.json";
|
||||
const OUTPUT_FILE = "output/notfound-payee-results.json";
|
||||
const PROGRESS_FILE = "output/notfound-payee-progress.json";
|
||||
const DELAY_MS = 800;
|
||||
|
||||
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
|
||||
console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`);
|
||||
|
||||
let completed = {};
|
||||
if (existsSync(PROGRESS_FILE)) {
|
||||
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
|
||||
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
|
||||
}
|
||||
|
||||
const remaining = allPOs.filter((po) => !completed[po]);
|
||||
console.log(`Remaining: ${remaining.length} POs\n`);
|
||||
|
||||
if (remaining.length === 0) {
|
||||
console.log("All POs already scraped!");
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
function poUrl(po) {
|
||||
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
|
||||
}
|
||||
|
||||
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
|
||||
const context = await chromium.launchPersistentContext(userDataDir, {
|
||||
headless: false,
|
||||
channel: "chrome",
|
||||
args: ["--disable-blink-features=AutomationControlled"],
|
||||
ignoreDefaultArgs: ["--enable-automation"],
|
||||
});
|
||||
|
||||
const page = context.pages()[0] || (await context.newPage());
|
||||
|
||||
console.log("Opening Payee Central...");
|
||||
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
|
||||
|
||||
if (page.url().includes("signin") || page.url().includes("/ap/")) {
|
||||
console.log("\n🔐 Login required. Please sign in in the browser window...");
|
||||
await page.waitForURL(
|
||||
(u) => {
|
||||
const s = u.toString();
|
||||
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
|
||||
},
|
||||
{ timeout: 180000 }
|
||||
);
|
||||
console.log("✅ Logged in!\n");
|
||||
}
|
||||
|
||||
let results = [];
|
||||
if (existsSync(OUTPUT_FILE)) {
|
||||
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
|
||||
}
|
||||
|
||||
let scraped = 0;
|
||||
let errors = 0;
|
||||
let consecutiveErrors = 0;
|
||||
const startTime = Date.now();
|
||||
|
||||
for (const po of remaining) {
|
||||
try {
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
|
||||
|
||||
if (page.url().includes("signin") || page.url().includes("/ap/")) {
|
||||
console.log("\n🔐 Session expired. Please log in again...");
|
||||
await page.waitForURL(
|
||||
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
|
||||
{ timeout: 180000 }
|
||||
);
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
|
||||
}
|
||||
|
||||
await page.waitForTimeout(500);
|
||||
const bodyText = await page.innerText("body");
|
||||
|
||||
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
|
||||
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
|
||||
|
||||
let siteCode = null;
|
||||
if (shipTo) {
|
||||
const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/);
|
||||
if (m) siteCode = m[1];
|
||||
if (!siteCode) {
|
||||
const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/);
|
||||
if (m2) siteCode = m2[1];
|
||||
}
|
||||
if (!siteCode) {
|
||||
const m3 = shipTo.match(/LLC\s+([A-Z0-9]{3,5})\b/);
|
||||
if (m3) siteCode = m3[1];
|
||||
}
|
||||
if (!siteCode) {
|
||||
const m4 = shipTo.match(/^([A-Z0-9]{3,5})\s*-\s*Amazon/m);
|
||||
if (m4) siteCode = m4[1];
|
||||
}
|
||||
if (!siteCode) {
|
||||
const m5 = shipTo.match(/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/);
|
||||
if (m5) siteCode = m5[1];
|
||||
}
|
||||
}
|
||||
|
||||
const lineItems = await page.$$eval(
|
||||
"table tbody tr",
|
||||
(rows) =>
|
||||
rows.map((row) => {
|
||||
const cells = row.querySelectorAll("td");
|
||||
if (cells.length < 7) return null;
|
||||
return {
|
||||
description: cells[1]?.textContent?.trim() || null,
|
||||
need_by: cells[2]?.textContent?.trim() || null,
|
||||
quantity: cells[3]?.textContent?.trim() || null,
|
||||
unit: cells[4]?.textContent?.trim() || null,
|
||||
price: cells[5]?.textContent?.trim() || null,
|
||||
amount: cells[6]?.textContent?.trim() || null,
|
||||
};
|
||||
}).filter(Boolean)
|
||||
).catch(() => []);
|
||||
|
||||
const result = {
|
||||
po_number: po,
|
||||
site_code: siteCode,
|
||||
ship_to_raw: shipTo,
|
||||
line_items: lineItems,
|
||||
};
|
||||
|
||||
results.push(result);
|
||||
completed[po] = true;
|
||||
scraped++;
|
||||
consecutiveErrors = 0;
|
||||
|
||||
const elapsed = (Date.now() - startTime) / 1000;
|
||||
const rate = scraped / elapsed;
|
||||
const eta = Math.round((remaining.length - scraped) / rate);
|
||||
const etaMin = Math.floor(eta / 60);
|
||||
const etaSec = eta % 60;
|
||||
|
||||
const desc0 = lineItems[0]?.description?.substring(0, 40) || "no items";
|
||||
process.stdout.write(
|
||||
`\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ${desc0} | ETA: ${etaMin}m${etaSec}s `
|
||||
);
|
||||
|
||||
if (scraped % 10 === 0) {
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`\n [${po}] Error: ${err.message}`);
|
||||
errors++;
|
||||
consecutiveErrors++;
|
||||
if (consecutiveErrors > 10) {
|
||||
console.error("Too many consecutive errors, stopping.");
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
await page.waitForTimeout(DELAY_MS);
|
||||
}
|
||||
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
|
||||
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
|
||||
console.log(`Total in output: ${results.length} POs`);
|
||||
console.log(`Results saved to ${OUTPUT_FILE}`);
|
||||
|
||||
await context.close();
|
||||
159
scrape-payee.mjs
Normal file
159
scrape-payee.mjs
Normal file
|
|
@ -0,0 +1,159 @@
|
|||
import { chromium } from "playwright";
|
||||
import { readFileSync, writeFileSync, existsSync } from "fs";
|
||||
|
||||
const PO_LIST = "output/unknown-pos.json";
|
||||
const OUTPUT_FILE = "output/payee-results.json";
|
||||
const PROGRESS_FILE = "output/payee-progress.json";
|
||||
const DELAY_MS = 800;
|
||||
|
||||
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
|
||||
console.log(`Loaded ${allPOs.length} POs to scrape from Payee Central`);
|
||||
|
||||
let completed = {};
|
||||
if (existsSync(PROGRESS_FILE)) {
|
||||
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
|
||||
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
|
||||
}
|
||||
|
||||
const remaining = allPOs.filter((po) => !completed[po]);
|
||||
console.log(`Remaining: ${remaining.length} POs\n`);
|
||||
|
||||
if (remaining.length === 0) {
|
||||
console.log("All POs already scraped!");
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
function poUrl(po) {
|
||||
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
|
||||
}
|
||||
|
||||
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
|
||||
const context = await chromium.launchPersistentContext(userDataDir, {
|
||||
headless: false,
|
||||
channel: "chrome",
|
||||
args: ["--disable-blink-features=AutomationControlled"],
|
||||
ignoreDefaultArgs: ["--enable-automation"],
|
||||
});
|
||||
|
||||
const page = context.pages()[0] || (await context.newPage());
|
||||
|
||||
console.log("Opening Payee Central...");
|
||||
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
|
||||
|
||||
if (page.url().includes("signin") || page.url().includes("/ap/")) {
|
||||
console.log("\n🔐 Login required. Please sign in in the browser window...");
|
||||
await page.waitForURL(
|
||||
(u) => {
|
||||
const s = u.toString();
|
||||
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
|
||||
},
|
||||
{ timeout: 180000 }
|
||||
);
|
||||
console.log("✅ Logged in!\n");
|
||||
}
|
||||
|
||||
let results = [];
|
||||
if (existsSync(OUTPUT_FILE)) {
|
||||
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
|
||||
}
|
||||
|
||||
let scraped = 0;
|
||||
let errors = 0;
|
||||
let consecutiveErrors = 0;
|
||||
const startTime = Date.now();
|
||||
|
||||
for (const po of remaining) {
|
||||
try {
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
|
||||
|
||||
if (page.url().includes("signin") || page.url().includes("/ap/")) {
|
||||
console.log("\n🔐 Session expired. Please log in again...");
|
||||
await page.waitForURL(
|
||||
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
|
||||
{ timeout: 180000 }
|
||||
);
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
|
||||
}
|
||||
|
||||
await page.waitForTimeout(500);
|
||||
const bodyText = await page.innerText("body");
|
||||
|
||||
// Extract Ship To with site code
|
||||
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
|
||||
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
|
||||
|
||||
let siteCode = null;
|
||||
if (shipTo) {
|
||||
const m = shipTo.match(/\(([A-Z0-9]{3,5})\)/);
|
||||
if (m) siteCode = m[1];
|
||||
if (!siteCode) {
|
||||
const m2 = shipTo.match(/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/);
|
||||
if (m2) siteCode = m2[1];
|
||||
}
|
||||
}
|
||||
|
||||
// Extract line items from table
|
||||
const lineItems = await page.$$eval(
|
||||
"table tbody tr",
|
||||
(rows) =>
|
||||
rows.map((row) => {
|
||||
const cells = row.querySelectorAll("td");
|
||||
if (cells.length < 7) return null;
|
||||
return {
|
||||
description: cells[1]?.textContent?.trim() || null,
|
||||
need_by: cells[2]?.textContent?.trim() || null,
|
||||
quantity: cells[3]?.textContent?.trim() || null,
|
||||
unit: cells[4]?.textContent?.trim() || null,
|
||||
price: cells[5]?.textContent?.trim() || null,
|
||||
amount: cells[6]?.textContent?.trim() || null,
|
||||
};
|
||||
}).filter(Boolean)
|
||||
).catch(() => []);
|
||||
|
||||
const result = {
|
||||
po_number: po,
|
||||
site_code: siteCode,
|
||||
ship_to_raw: shipTo,
|
||||
line_items: lineItems,
|
||||
};
|
||||
|
||||
results.push(result);
|
||||
completed[po] = true;
|
||||
scraped++;
|
||||
consecutiveErrors = 0;
|
||||
|
||||
const elapsed = (Date.now() - startTime) / 1000;
|
||||
const rate = scraped / elapsed;
|
||||
const eta = Math.round((remaining.length - scraped) / rate);
|
||||
const etaMin = Math.floor(eta / 60);
|
||||
const etaSec = eta % 60;
|
||||
|
||||
process.stdout.write(
|
||||
`\r[${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s `
|
||||
);
|
||||
|
||||
if (scraped % 10 === 0) {
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`\n [${po}] Error: ${err.message}`);
|
||||
errors++;
|
||||
consecutiveErrors++;
|
||||
if (consecutiveErrors > 10) {
|
||||
console.error("Too many consecutive errors, stopping.");
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
await page.waitForTimeout(DELAY_MS);
|
||||
}
|
||||
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
|
||||
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
|
||||
console.log(`Total in output: ${results.length} POs`);
|
||||
console.log(`Results saved to ${OUTPUT_FILE}`);
|
||||
|
||||
await context.close();
|
||||
292
scrape.mjs
Normal file
292
scrape.mjs
Normal file
|
|
@ -0,0 +1,292 @@
|
|||
import { chromium } from "playwright";
|
||||
import { readFileSync, writeFileSync, existsSync } from "fs";
|
||||
|
||||
const EMAIL = process.env.COUPA_EMAIL;
|
||||
const PASSWORD = process.env.COUPA_PASSWORD;
|
||||
|
||||
if (!EMAIL || !PASSWORD) {
|
||||
console.error("Set COUPA_EMAIL and COUPA_PASSWORD env vars");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const SUPPLIER_ID = "895025";
|
||||
const INSTANCE_URL = "https%3A%2F%2Famazon.coupahost.com%2F";
|
||||
const OUTPUT_FILE = "output/scraped-pos.json";
|
||||
const PROGRESS_FILE = "output/progress.json";
|
||||
const DELAY_MS = 1500;
|
||||
const KEEPALIVE_EVERY = 75;
|
||||
|
||||
const poListFile = process.argv[2];
|
||||
if (!poListFile) {
|
||||
console.error("Usage: node scrape.mjs <po-list.json>");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const allPOs = JSON.parse(readFileSync(poListFile, "utf-8"));
|
||||
console.log(`Loaded ${allPOs.length} POs to scrape`);
|
||||
|
||||
let completed = {};
|
||||
if (existsSync(PROGRESS_FILE)) {
|
||||
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
|
||||
console.log(`Resuming: ${Object.keys(completed).length} already scraped`);
|
||||
}
|
||||
|
||||
const remaining = allPOs.filter((po) => !completed[po]);
|
||||
console.log(`Remaining: ${remaining.length} POs\n`);
|
||||
|
||||
if (remaining.length === 0) {
|
||||
console.log("All POs already scraped!");
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
function poUrl(po) {
|
||||
const numeric = po.split("-").slice(1).join("-");
|
||||
return `https://supplier.coupahost.com/orders/${numeric}?instance_url=${INSTANCE_URL}&supplier_id=${SUPPLIER_ID}&`;
|
||||
}
|
||||
|
||||
function isLoginPage(url) {
|
||||
return url.includes("sessions/new") || url.includes("login") || url.includes("sign_in");
|
||||
}
|
||||
|
||||
async function autoLogin(page) {
|
||||
console.log("\n🔐 Session expired — auto-filling login form...");
|
||||
|
||||
// Step 1: Fill email and click Continue
|
||||
try {
|
||||
await page.waitForSelector("input#email", { state: "visible", timeout: 5000 });
|
||||
await page.fill("input#email", EMAIL);
|
||||
await page.click("button.s-login").catch(() => {});
|
||||
console.log(" Email filled, clicked Continue.");
|
||||
} catch {
|
||||
// Already past email step
|
||||
}
|
||||
|
||||
// Step 2: CAPTCHA appears before password — wait for user to solve it
|
||||
// After CAPTCHA, password field will appear
|
||||
console.log(" ⏳ Solve the CAPTCHA — password will auto-fill after...");
|
||||
try {
|
||||
await page.waitForSelector("input#password", { state: "visible", timeout: 120000 });
|
||||
await page.fill("input#password", PASSWORD);
|
||||
console.log(" Password filled, clicking Login...");
|
||||
page.click("button.s-login").catch(() => {});
|
||||
} catch {
|
||||
console.log(" Password field never appeared. Check the browser.");
|
||||
}
|
||||
|
||||
// Wait for login to complete (may need another CAPTCHA or just redirects)
|
||||
console.log(" ⏳ Waiting for login to complete...");
|
||||
try {
|
||||
await page.waitForURL((url) => !isLoginPage(url.toString()), { timeout: 120000 });
|
||||
console.log(" ✅ Login successful!\n");
|
||||
return true;
|
||||
} catch {
|
||||
console.error(" ❌ Login timed out after 120s. Exiting.");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
async function keepAlive(page) {
|
||||
try {
|
||||
await page.goto("https://supplier.coupahost.com/home/", {
|
||||
waitUntil: "domcontentloaded",
|
||||
timeout: 15000,
|
||||
});
|
||||
if (isLoginPage(page.url())) return false;
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
async function scrapePO(page, po) {
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 30000 });
|
||||
|
||||
if (isLoginPage(page.url())) {
|
||||
const ok = await autoLogin(page);
|
||||
if (!ok) return null;
|
||||
// Navigate to PO after re-login
|
||||
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 30000 });
|
||||
if (isLoginPage(page.url())) return null;
|
||||
}
|
||||
|
||||
await page.waitForSelector("iframe#enterprise_frame", { timeout: 15000 });
|
||||
await page.waitForTimeout(1500);
|
||||
|
||||
const frame =
|
||||
page.frame({ name: "enterprise_frame" }) ||
|
||||
page.frames().find((f) => f.url().includes("supplier_order_headers"));
|
||||
|
||||
if (!frame) return "no_iframe";
|
||||
|
||||
await frame.waitForLoadState("networkidle");
|
||||
|
||||
const data = await frame.evaluate(() => {
|
||||
const text = (sel) => {
|
||||
const el = document.querySelector(sel);
|
||||
return el ? el.textContent.trim() : null;
|
||||
};
|
||||
|
||||
const lineItems = [];
|
||||
const rows = document.querySelectorAll(".coupa_datatable_row");
|
||||
for (const row of rows) {
|
||||
const desc = row.querySelector(".s-description");
|
||||
const price = row.querySelector(".s-price span[title]");
|
||||
const total = row.querySelector(".s-total span[title]");
|
||||
const needBy = row.querySelector("[id$='_need_by_date']");
|
||||
lineItems.push({
|
||||
description: desc ? desc.textContent.trim() : null,
|
||||
price: price ? price.textContent.trim() : null,
|
||||
total: total ? total.textContent.trim() : null,
|
||||
need_by: needBy ? needBy.textContent.trim() : null,
|
||||
});
|
||||
}
|
||||
|
||||
const addrEl = document.querySelector(
|
||||
"label[for='order_header_ship_to_address'] + span.address, span.address"
|
||||
);
|
||||
let shipTo = null;
|
||||
if (addrEl) {
|
||||
const clone = addrEl.cloneNode(true);
|
||||
clone.querySelectorAll(".form_element").forEach((el) => el.remove());
|
||||
shipTo = clone.innerHTML
|
||||
.replace(/<br\s*\/?>/gi, "\n")
|
||||
.replace(/<[^>]+>/g, "")
|
||||
.trim();
|
||||
}
|
||||
|
||||
return {
|
||||
status: text(".s-readable_status"),
|
||||
order_date: text(".s-local_header_created_at"),
|
||||
requester: text(".s-requester"),
|
||||
payment_term: text("#order_header_payment_term .data"),
|
||||
ship_to_raw: shipTo,
|
||||
line_items: lineItems,
|
||||
};
|
||||
});
|
||||
|
||||
let site_code = null;
|
||||
if (data.ship_to_raw) {
|
||||
const m = data.ship_to_raw.match(/\(([A-Z0-9]{3,5})\)/);
|
||||
if (m) site_code = m[1];
|
||||
if (!site_code) {
|
||||
const m2 = data.ship_to_raw.match(/Attn:\s*([A-Z0-9]{3,5})\b/);
|
||||
if (m2) site_code = m2[1];
|
||||
}
|
||||
}
|
||||
|
||||
return { po_number: po, site_code, ...data };
|
||||
}
|
||||
|
||||
// --- Main ---
|
||||
// Use a persistent Chrome profile so reCAPTCHA trusts the browser
|
||||
const userDataDir = new URL("./chrome-profile", import.meta.url).pathname;
|
||||
const context = await chromium.launchPersistentContext(userDataDir, {
|
||||
headless: false,
|
||||
channel: "chrome",
|
||||
args: ["--disable-blink-features=AutomationControlled"],
|
||||
ignoreDefaultArgs: ["--enable-automation"],
|
||||
});
|
||||
const page = context.pages()[0] || (await context.newPage());
|
||||
|
||||
// Initial login
|
||||
console.log("Opening Coupa supplier portal...");
|
||||
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
|
||||
|
||||
if (isLoginPage(page.url())) {
|
||||
const ok = await autoLogin(page);
|
||||
if (!ok) {
|
||||
await context.close();
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
let results = [];
|
||||
if (existsSync(OUTPUT_FILE)) {
|
||||
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
|
||||
}
|
||||
|
||||
let scraped = 0;
|
||||
let errors = 0;
|
||||
let consecutiveErrors = 0;
|
||||
const startTime = Date.now();
|
||||
|
||||
for (const po of remaining) {
|
||||
// Session keep-alive
|
||||
if (scraped > 0 && scraped % KEEPALIVE_EVERY === 0) {
|
||||
process.stdout.write("\n Refreshing session...");
|
||||
const alive = await keepAlive(page);
|
||||
if (!alive) {
|
||||
console.log(" session expired, re-logging in...");
|
||||
await page.goto("https://supplier.coupahost.com/sessions/new", {
|
||||
waitUntil: "networkidle",
|
||||
});
|
||||
const ok = await autoLogin(page);
|
||||
if (!ok) break;
|
||||
} else {
|
||||
process.stdout.write(" ok\n");
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const result = await scrapePO(page, po);
|
||||
|
||||
if (result === null) {
|
||||
console.error(`\nFailed to authenticate. Saving progress.`);
|
||||
break;
|
||||
}
|
||||
|
||||
if (result === "no_iframe") {
|
||||
console.error(`\n [${po}] No iframe found, skipping`);
|
||||
errors++;
|
||||
consecutiveErrors++;
|
||||
if (consecutiveErrors > 10) {
|
||||
console.error("Too many consecutive errors, stopping.");
|
||||
break;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
results.push(result);
|
||||
completed[po] = true;
|
||||
scraped++;
|
||||
consecutiveErrors = 0;
|
||||
|
||||
const elapsed = (Date.now() - startTime) / 1000;
|
||||
const rate = scraped / elapsed;
|
||||
const eta = Math.round((remaining.length - scraped) / rate);
|
||||
const etaMin = Math.floor(eta / 60);
|
||||
const etaSec = eta % 60;
|
||||
const desc0 = result.line_items[0]?.description?.substring(0, 50) || "no lines";
|
||||
|
||||
process.stdout.write(
|
||||
`\r[${scraped}/${remaining.length}] ${po} | ${result.site_code || "?"} | ${desc0}... | ETA: ${etaMin}m${etaSec}s `
|
||||
);
|
||||
|
||||
// Save every 10 POs (more frequent for monitor)
|
||||
if (scraped % 10 === 0) {
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`\n [${po}] Error: ${err.message}`);
|
||||
errors++;
|
||||
consecutiveErrors++;
|
||||
|
||||
if (consecutiveErrors > 10) {
|
||||
console.error("Too many consecutive errors, stopping.");
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
await page.waitForTimeout(DELAY_MS);
|
||||
}
|
||||
|
||||
// Final save
|
||||
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
|
||||
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
|
||||
|
||||
console.log(`\n\nDone! Scraped: ${scraped}, Errors: ${errors}`);
|
||||
console.log(`Total in output: ${results.length} POs`);
|
||||
console.log(`Results saved to ${OUTPUT_FILE}`);
|
||||
|
||||
await context.close();
|
||||
Reference in a new issue