#!/usr/bin/env python3.12 """Parse Coupa PO emails from an MBOX file and extract line items + ship-to data.""" import mailbox import email import re import json import sys import os def parse_amount(s): """Parse amount string, handling formats like '1.0 EACH x 3,000.00' or '10,000.00'.""" s = s.strip() if ' x ' in s.lower(): s = s.split(' x ')[-1].strip() elif ' X ' in s: s = s.split(' X ')[-1].strip() return float(s.replace(',', '')) def parse_po_email(body): """Extract PO data from a Coupa 'issued' email plaintext body.""" lines = body.split('\n') lines = [l.strip() for l in lines] # PO number from body po_match = re.search(r'Purchase Order #?((?:2D|B187|FK)-\d+)', body) if not po_match: return None po_number = po_match.group(1) # Line items: between "Items" line and the coupahost URL items_start = None items_end = None for i, line in enumerate(lines): if line == 'Items' and items_start is None: items_start = i + 1 if items_start and 'supplier.coupahost.com/orders/' in line: items_end = i break line_items = [] if items_start and items_end: item_lines = [l for l in lines[items_start:items_end] if l] i = 0 while i < len(item_lines): desc = item_lines[i] # Skip if this line looks like a number/currency if re.match(r'^[\d,]', desc) or desc == 'USD': i += 1 continue amount = None # Look ahead for amount for j in range(i + 1, min(i + 4, len(item_lines))): candidate = item_lines[j] if re.match(r'^[\d,]', candidate): try: amount = parse_amount(candidate) except ValueError: pass break line_items.append({ 'description': desc, 'amount': amount, }) # Skip past amount + currency lines i = j + 1 if amount is not None else i + 1 while i < len(item_lines) and item_lines[i] == 'USD': i += 1 # Key-value pairs after "More Detail" kv = {} detail_start = None for i, line in enumerate(lines): if line == 'More Detail': detail_start = i + 1 break if detail_start: kv_keys = ['PO ID', 'Department', 'Status', 'Last Opened', 'Order Date', 'Acknowledged At', 'Revision Date', 'Payment Term', 'Req #'] for i in range(detail_start, len(lines)): for k in kv_keys: if lines[i] == k and i + 1 < len(lines): kv[k] = lines[i + 1] # Ship-to address: second "Shipping" section shipping_count = 0 ship_to_lines = [] capturing = False skipping_blanks = False for i, line in enumerate(lines): if line == 'Shipping': shipping_count += 1 if shipping_count == 2: capturing = True skipping_blanks = True continue elif capturing: if skipping_blanks and line == '': continue skipping_blanks = False if line in ('', 'Ship To Address') or line.startswith('---') or line.startswith('http'): break if line in kv_keys or line == 'Supplier' or line == 'More Detail': break ship_to_lines.append(line) ship_to_raw = '\n'.join(ship_to_lines) if ship_to_lines else None # Extract site code from ship-to site_code = None if ship_to_raw: # (CODE) pattern — e.g. "Amazon.com Services LLC (XSF2)" m = re.search(r'\(([A-Z0-9]{3,5})\)', ship_to_raw) if m: site_code = m.group(1) # "LLC - CODE" — e.g. "Amazon.com Services LLC - WUT9" if not site_code: m = re.search(r'(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b', ship_to_raw) if m: site_code = m.group(1) # "ATTN: ... Station CODE" or "ATTN: ... DS - CODE" if not site_code: m = re.search(r'ATTN:.*?(?:Station|DS)\s*[-–]?\s*([A-Z0-9]{3,5})\b', ship_to_raw) if m: site_code = m.group(1) # "CODE - Amazon" at start of first line if not site_code: m = re.match(r'^([A-Z0-9]{3,5})\s*-\s*Amazon', ship_to_raw) if m: site_code = m.group(1) if not site_code and line_items: for li in line_items: m = re.match(r'^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b', li['description']) if m: site_code = m.group(1) break return { 'po_number': po_number, 'site_code': site_code, 'status': kv.get('Status'), 'order_date': kv.get('Order Date'), 'revision_date': kv.get('Revision Date'), 'payment_term': kv.get('Payment Term'), 'req_number': kv.get('Req #'), 'ship_to_raw': ship_to_raw, 'line_items': line_items, } def process_mbox(mbox_path, target_pos=None, output_path=None): """Process an MBOX file and return parsed PO records.""" mbox = mailbox.mbox(mbox_path) results = {} processed = 0 matched = 0 for msg in mbox: processed += 1 subject = msg.get('Subject', '') if 'issued' not in subject.lower() or 'Purchase Order' not in subject: continue if msg.is_multipart(): body = None for part in msg.walk(): if part.get_content_type() == 'text/plain': body = part.get_payload(decode=True).decode('utf-8', errors='replace') break if not body: continue else: body = msg.get_payload(decode=True).decode('utf-8', errors='replace') record = parse_po_email(body) if not record: continue po = record['po_number'] if target_pos and po not in target_pos: continue # Keep latest version if duplicate if po not in results: results[po] = record matched += 1 if matched % 500 == 0 and matched > 0: print(f" Matched {matched} POs so far... ({processed} messages scanned)", file=sys.stderr) mbox.close() result_list = list(results.values()) print(f"Processed {processed} messages, matched {matched} unique POs", file=sys.stderr) if output_path: with open(output_path, 'w') as f: json.dump(result_list, f, indent=2) print(f"Saved to {output_path}", file=sys.stderr) return result_list if __name__ == '__main__': mbox_path = sys.argv[1] if len(sys.argv) > 1 else '/tmp/coupa-emails/coupa-po-dump--info@seahavenind.com-wHxmHD.mbox' po_list_path = sys.argv[2] if len(sys.argv) > 2 else 'output/po-list.json' output_path = sys.argv[3] if len(sys.argv) > 3 else 'output/email-parsed-pos.json' target_pos = None if os.path.exists(po_list_path): with open(po_list_path) as f: target_pos = set(json.load(f)) print(f"Targeting {len(target_pos)} specific POs", file=sys.stderr) process_mbox(mbox_path, target_pos, output_path)