This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
amazon-po-parser/scrape-fill-chunk.mjs
Adam Moussa a6bb1adfbc Add reusable chunk scraper, LLM prompt, README; clean up one-off scripts
- Add scrape-fill-chunk.mjs: reusable chunk-based Payee Central scraper
  with resume support and concurrent instance capability
- Add prompts/po-email-parser.md: LLM prompt for Coupa PO email parsing
  with site code extraction patterns, skip list, and trade classification
- Add README.md with architecture docs, DynamoDB schema, and script index
- Track output/site-state-extra-mapping.json (392 facility codes)
- Remove scrape-payee.mjs and scrape-payee-notfound.mjs (superseded by
  chunk scraper)
2026-05-01 19:52:07 -04:00

151 lines
4.8 KiB
JavaScript

import { chromium } from "playwright";
import { readFileSync, writeFileSync, existsSync } from "fs";
import { execSync } from "child_process";
const chunkId = process.argv[2];
if (!chunkId && chunkId !== "0") {
console.error("Usage: node scrape-fill-chunk.mjs <chunk-id>");
process.exit(1);
}
const PO_LIST = `output/fill-chunk${chunkId}.json`;
const OUTPUT_FILE = `output/fill-chunk${chunkId}-results.json`;
const PROGRESS_FILE = `output/fill-chunk${chunkId}-progress.json`;
const PROFILE_DIR = `chrome-profile-fill-${chunkId}`;
const DELAY_MS = 800;
const allPOs = JSON.parse(readFileSync(PO_LIST, "utf-8"));
console.log(`[Chunk ${chunkId}] Loaded ${allPOs.length} POs`);
let completed = {};
if (existsSync(PROGRESS_FILE)) {
completed = JSON.parse(readFileSync(PROGRESS_FILE, "utf-8"));
console.log(`[Chunk ${chunkId}] Resuming: ${Object.keys(completed).length} already done`);
}
const remaining = allPOs.filter((po) => !completed[po]);
console.log(`[Chunk ${chunkId}] Remaining: ${remaining.length} POs\n`);
if (remaining.length === 0) {
console.log("All POs already scraped!");
process.exit(0);
}
function poUrl(po) {
return `https://payeecentral.amazon.com/PurchaseOrder/Details?poNumber=${po}`;
}
const profilePath = new URL(`./${PROFILE_DIR}`, import.meta.url).pathname;
const mainProfile = new URL("./chrome-profile", import.meta.url).pathname;
if (!existsSync(profilePath)) {
console.log(`Copying chrome profile to ${PROFILE_DIR}...`);
execSync(`cp -R "${mainProfile}" "${profilePath}"`);
}
const context = await chromium.launchPersistentContext(profilePath, {
headless: false,
channel: "chrome",
args: ["--disable-blink-features=AutomationControlled"],
ignoreDefaultArgs: ["--enable-automation"],
});
const page = context.pages()[0] || (await context.newPage());
console.log(`[Chunk ${chunkId}] Opening Payee Central...`);
await page.goto(poUrl(remaining[0]), { waitUntil: "networkidle", timeout: 30000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log(`\n[Chunk ${chunkId}] Login required. Please sign in...`);
await page.waitForURL(
(u) => {
const s = u.toString();
return s.includes("payeecentral.amazon.com") && !s.includes("signin") && !s.includes("/ap/");
},
{ timeout: 180000 }
);
console.log(`[Chunk ${chunkId}] Logged in!\n`);
}
let results = [];
if (existsSync(OUTPUT_FILE)) {
results = JSON.parse(readFileSync(OUTPUT_FILE, "utf-8"));
}
let scraped = 0;
let errors = 0;
let consecutiveErrors = 0;
const startTime = Date.now();
for (const po of remaining) {
try {
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
if (page.url().includes("signin") || page.url().includes("/ap/")) {
console.log(`\n[Chunk ${chunkId}] Session expired. Please log in again...`);
await page.waitForURL(
(u) => !u.toString().includes("signin") && !u.toString().includes("/ap/"),
{ timeout: 180000 }
);
await page.goto(poUrl(po), { waitUntil: "networkidle", timeout: 20000 });
}
await page.waitForTimeout(500);
const bodyText = await page.innerText("body");
const shipToMatch = bodyText.match(/Ship To\s+([\s\S]*?)(?=Payee Central Account)/);
const shipTo = shipToMatch ? shipToMatch[1].trim() : null;
let siteCode = null;
if (shipTo) {
for (const re of [
/\(([A-Z0-9]{3,5})\)/,
/(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b/,
/LLC\s+([A-Z0-9]{3,5})\b/,
/^([A-Z0-9]{3,5})\s*-\s*Amazon/m,
/Amazon\s+Fresh\s+([A-Z0-9]{3,5})\b/,
]) {
const m = shipTo.match(re);
if (m) { siteCode = m[1]; break; }
}
}
results.push({
po_number: po,
site_code: siteCode,
ship_to_raw: shipTo,
});
completed[po] = true;
scraped++;
consecutiveErrors = 0;
const elapsed = (Date.now() - startTime) / 1000;
const rate = scraped / elapsed;
const eta = Math.round((remaining.length - scraped) / rate);
const etaMin = Math.floor(eta / 60);
const etaSec = eta % 60;
process.stdout.write(
`\r[C${chunkId}] [${scraped}/${remaining.length}] ${po} | ${siteCode || "?"} | ETA: ${etaMin}m${etaSec}s `
);
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
} catch (err) {
console.error(`\n[C${chunkId}] [${po}] Error: ${err.message}`);
errors++;
consecutiveErrors++;
if (consecutiveErrors > 10) {
console.error("Too many consecutive errors, stopping.");
break;
}
}
await page.waitForTimeout(DELAY_MS);
}
writeFileSync(OUTPUT_FILE, JSON.stringify(results, null, 2));
writeFileSync(PROGRESS_FILE, JSON.stringify(completed));
console.log(`\n\n[Chunk ${chunkId}] Done! Scraped: ${scraped}, Errors: ${errors}`);
await context.close();