From 702c65541a0cd2ed8d264e57486cf32b23c05178 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Fri, 8 May 2026 15:45:52 -0400 Subject: [PATCH] Add CI workflow and apply ruff formatting --- .github/workflows/ci.yaml | 11 +++ parse-mbox.py | 136 +++++++++++++++++++++++--------------- 2 files changed, 93 insertions(+), 54 deletions(-) create mode 100644 .github/workflows/ci.yaml diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml new file mode 100644 index 0000000..5e672a6 --- /dev/null +++ b/.github/workflows/ci.yaml @@ -0,0 +1,11 @@ +name: CI +on: + pull_request: + branches: [main] + +jobs: + ci: + uses: Sea-Haven-Industries/.github/.github/workflows/ci-python-sam.yaml@main + with: + source-dirs: "." + run-sam-validate: false diff --git a/parse-mbox.py b/parse-mbox.py index 26cdcb0..ae98873 100644 --- a/parse-mbox.py +++ b/parse-mbox.py @@ -2,29 +2,29 @@ """Parse Coupa PO emails from an MBOX file and extract line items + ship-to data.""" import mailbox -import email import re import json import sys import os + def parse_amount(s): """Parse amount string, handling formats like '1.0 EACH x 3,000.00' or '10,000.00'.""" s = s.strip() - if ' x ' in s.lower(): - s = s.split(' x ')[-1].strip() - elif ' X ' in s: - s = s.split(' X ')[-1].strip() - return float(s.replace(',', '')) + if " x " in s.lower(): + s = s.split(" x ")[-1].strip() + elif " X " in s: + s = s.split(" X ")[-1].strip() + return float(s.replace(",", "")) def parse_po_email(body): """Extract PO data from a Coupa 'issued' email plaintext body.""" - lines = body.split('\n') - lines = [l.strip() for l in lines] + lines = body.split("\n") + lines = [line.strip() for line in lines] # PO number from body - po_match = re.search(r'Purchase Order #?((?:2D|B187|FK)-\d+)', body) + po_match = re.search(r"Purchase Order #?((?:2D|B187|FK)-\d+)", body) if not po_match: return None po_number = po_match.group(1) @@ -33,52 +33,63 @@ def parse_po_email(body): items_start = None items_end = None for i, line in enumerate(lines): - if line == 'Items' and items_start is None: + if line == "Items" and items_start is None: items_start = i + 1 - if items_start and 'supplier.coupahost.com/orders/' in line: + if items_start and "supplier.coupahost.com/orders/" in line: items_end = i break line_items = [] if items_start and items_end: - item_lines = [l for l in lines[items_start:items_end] if l] + item_lines = [line for line in lines[items_start:items_end] if line] i = 0 while i < len(item_lines): desc = item_lines[i] # Skip if this line looks like a number/currency - if re.match(r'^[\d,]', desc) or desc == 'USD': + if re.match(r"^[\d,]", desc) or desc == "USD": i += 1 continue amount = None # Look ahead for amount for j in range(i + 1, min(i + 4, len(item_lines))): candidate = item_lines[j] - if re.match(r'^[\d,]', candidate): + if re.match(r"^[\d,]", candidate): try: amount = parse_amount(candidate) except ValueError: pass break - line_items.append({ - 'description': desc, - 'amount': amount, - }) + line_items.append( + { + "description": desc, + "amount": amount, + } + ) # Skip past amount + currency lines i = j + 1 if amount is not None else i + 1 - while i < len(item_lines) and item_lines[i] == 'USD': + while i < len(item_lines) and item_lines[i] == "USD": i += 1 # Key-value pairs after "More Detail" kv = {} detail_start = None for i, line in enumerate(lines): - if line == 'More Detail': + if line == "More Detail": detail_start = i + 1 break if detail_start: - kv_keys = ['PO ID', 'Department', 'Status', 'Last Opened', 'Order Date', - 'Acknowledged At', 'Revision Date', 'Payment Term', 'Req #'] + kv_keys = [ + "PO ID", + "Department", + "Status", + "Last Opened", + "Order Date", + "Acknowledged At", + "Revision Date", + "Payment Term", + "Req #", + ] for i in range(detail_start, len(lines)): for k in kv_keys: if lines[i] == k and i + 1 < len(lines): @@ -90,63 +101,69 @@ def parse_po_email(body): capturing = False skipping_blanks = False for i, line in enumerate(lines): - if line == 'Shipping': + if line == "Shipping": shipping_count += 1 if shipping_count == 2: capturing = True skipping_blanks = True continue elif capturing: - if skipping_blanks and line == '': + if skipping_blanks and line == "": continue skipping_blanks = False - if line in ('', 'Ship To Address') or line.startswith('---') or line.startswith('http'): + if ( + line in ("", "Ship To Address") + or line.startswith("---") + or line.startswith("http") + ): break - if line in kv_keys or line == 'Supplier' or line == 'More Detail': + if line in kv_keys or line == "Supplier" or line == "More Detail": break ship_to_lines.append(line) - ship_to_raw = '\n'.join(ship_to_lines) if ship_to_lines else None + ship_to_raw = "\n".join(ship_to_lines) if ship_to_lines else None # Extract site code from ship-to site_code = None if ship_to_raw: # (CODE) pattern — e.g. "Amazon.com Services LLC (XSF2)" - m = re.search(r'\(([A-Z0-9]{3,5})\)', ship_to_raw) + m = re.search(r"\(([A-Z0-9]{3,5})\)", ship_to_raw) if m: site_code = m.group(1) # "LLC - CODE" — e.g. "Amazon.com Services LLC - WUT9" if not site_code: - m = re.search(r'(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b', ship_to_raw) + m = re.search(r"(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b", ship_to_raw) if m: site_code = m.group(1) # "ATTN: ... Station CODE" or "ATTN: ... DS - CODE" if not site_code: - m = re.search(r'ATTN:.*?(?:Station|DS)\s*[-–]?\s*([A-Z0-9]{3,5})\b', ship_to_raw) + m = re.search( + r"ATTN:.*?(?:Station|DS)\s*[-–]?\s*([A-Z0-9]{3,5})\b", ship_to_raw + ) if m: site_code = m.group(1) # "CODE - Amazon" at start of first line if not site_code: - m = re.match(r'^([A-Z0-9]{3,5})\s*-\s*Amazon', ship_to_raw) + m = re.match(r"^([A-Z0-9]{3,5})\s*-\s*Amazon", ship_to_raw) if m: site_code = m.group(1) if not site_code and line_items: for li in line_items: - m = re.match(r'^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b', li['description']) + m = re.match(r"^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b", li["description"]) if m: site_code = m.group(1) break return { - 'po_number': po_number, - 'site_code': site_code, - 'status': kv.get('Status'), - 'order_date': kv.get('Order Date'), - 'revision_date': kv.get('Revision Date'), - 'payment_term': kv.get('Payment Term'), - 'req_number': kv.get('Req #'), - 'ship_to_raw': ship_to_raw, - 'line_items': line_items, + "po_number": po_number, + "site_code": site_code, + "status": kv.get("Status"), + "order_date": kv.get("Order Date"), + "revision_date": kv.get("Revision Date"), + "payment_term": kv.get("Payment Term"), + "req_number": kv.get("Req #"), + "ship_to_raw": ship_to_raw, + "line_items": line_items, } @@ -159,27 +176,29 @@ def process_mbox(mbox_path, target_pos=None, output_path=None): for msg in mbox: processed += 1 - subject = msg.get('Subject', '') + subject = msg.get("Subject", "") - if 'issued' not in subject.lower() or 'Purchase Order' not in subject: + if "issued" not in subject.lower() or "Purchase Order" not in subject: continue if msg.is_multipart(): body = None for part in msg.walk(): - if part.get_content_type() == 'text/plain': - body = part.get_payload(decode=True).decode('utf-8', errors='replace') + if part.get_content_type() == "text/plain": + body = part.get_payload(decode=True).decode( + "utf-8", errors="replace" + ) break if not body: continue else: - body = msg.get_payload(decode=True).decode('utf-8', errors='replace') + body = msg.get_payload(decode=True).decode("utf-8", errors="replace") record = parse_po_email(body) if not record: continue - po = record['po_number'] + po = record["po_number"] if target_pos and po not in target_pos: continue @@ -190,25 +209,34 @@ def process_mbox(mbox_path, target_pos=None, output_path=None): matched += 1 if matched % 500 == 0 and matched > 0: - print(f" Matched {matched} POs so far... ({processed} messages scanned)", file=sys.stderr) + print( + f" Matched {matched} POs so far... ({processed} messages scanned)", + file=sys.stderr, + ) mbox.close() result_list = list(results.values()) - print(f"Processed {processed} messages, matched {matched} unique POs", file=sys.stderr) + print( + f"Processed {processed} messages, matched {matched} unique POs", file=sys.stderr + ) if output_path: - with open(output_path, 'w') as f: + with open(output_path, "w") as f: json.dump(result_list, f, indent=2) print(f"Saved to {output_path}", file=sys.stderr) return result_list -if __name__ == '__main__': - mbox_path = sys.argv[1] if len(sys.argv) > 1 else '/tmp/coupa-emails/coupa-po-dump--info@seahavenind.com-wHxmHD.mbox' - po_list_path = sys.argv[2] if len(sys.argv) > 2 else 'output/po-list.json' - output_path = sys.argv[3] if len(sys.argv) > 3 else 'output/email-parsed-pos.json' +if __name__ == "__main__": + mbox_path = ( + sys.argv[1] + if len(sys.argv) > 1 + else "/tmp/coupa-emails/coupa-po-dump--info@seahavenind.com-wHxmHD.mbox" + ) + po_list_path = sys.argv[2] if len(sys.argv) > 2 else "output/po-list.json" + output_path = sys.argv[3] if len(sys.argv) > 3 else "output/email-parsed-pos.json" target_pos = None if os.path.exists(po_list_path):