Add CI workflow and apply ruff formatting (#10)

This commit is contained in:
Adam Moussa 2026-05-08 15:56:24 -04:00 • committed by GitHub
parent 9eefbc0c73
commit 08b17db012
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 93 additions and 54 deletions

11
.github/workflows/ci.yaml vendored Normal file
View file

@ -0,0 +1,11 @@
name: CI
on:
pull_request:
branches: [main]
jobs:
ci:
uses: Sea-Haven-Industries/.github/.github/workflows/ci-python-sam.yaml@main
with:
source-dirs: "."
run-sam-validate: false

View file

@ -2,29 +2,29 @@
"""Parse Coupa PO emails from an MBOX file and extract line items + ship-to data.""" """Parse Coupa PO emails from an MBOX file and extract line items + ship-to data."""
import mailbox import mailbox
import email
import re import re
import json import json
import sys import sys
import os import os
def parse_amount(s): def parse_amount(s):
"""Parse amount string, handling formats like '1.0 EACH x 3,000.00' or '10,000.00'.""" """Parse amount string, handling formats like '1.0 EACH x 3,000.00' or '10,000.00'."""
s = s.strip() s = s.strip()
if ' x ' in s.lower(): if " x " in s.lower():
s = s.split(' x ')[-1].strip() s = s.split(" x ")[-1].strip()
elif ' X ' in s: elif " X " in s:
s = s.split(' X ')[-1].strip() s = s.split(" X ")[-1].strip()
return float(s.replace(',', '')) return float(s.replace(",", ""))
def parse_po_email(body): def parse_po_email(body):
"""Extract PO data from a Coupa 'issued' email plaintext body.""" """Extract PO data from a Coupa 'issued' email plaintext body."""
lines = body.split('\n') lines = body.split("\n")
lines = [l.strip() for l in lines] lines = [line.strip() for line in lines]
# PO number from body # PO number from body
po_match = re.search(r'Purchase Order #?((?:2D|B187|FK)-\d+)', body) po_match = re.search(r"Purchase Order #?((?:2D|B187|FK)-\d+)", body)
if not po_match: if not po_match:
return None return None
po_number = po_match.group(1) po_number = po_match.group(1)
@ -33,52 +33,63 @@ def parse_po_email(body):
items_start = None items_start = None
items_end = None items_end = None
for i, line in enumerate(lines): for i, line in enumerate(lines):
if line == 'Items' and items_start is None: if line == "Items" and items_start is None:
items_start = i + 1 items_start = i + 1
if items_start and 'supplier.coupahost.com/orders/' in line: if items_start and "supplier.coupahost.com/orders/" in line:
items_end = i items_end = i
break break
line_items = [] line_items = []
if items_start and items_end: if items_start and items_end:
item_lines = [l for l in lines[items_start:items_end] if l] item_lines = [line for line in lines[items_start:items_end] if line]
i = 0 i = 0
while i < len(item_lines): while i < len(item_lines):
desc = item_lines[i] desc = item_lines[i]
# Skip if this line looks like a number/currency # Skip if this line looks like a number/currency
if re.match(r'^[\d,]', desc) or desc == 'USD': if re.match(r"^[\d,]", desc) or desc == "USD":
i += 1 i += 1
continue continue
amount = None amount = None
# Look ahead for amount # Look ahead for amount
for j in range(i + 1, min(i + 4, len(item_lines))): for j in range(i + 1, min(i + 4, len(item_lines))):
candidate = item_lines[j] candidate = item_lines[j]
if re.match(r'^[\d,]', candidate): if re.match(r"^[\d,]", candidate):
try: try:
amount = parse_amount(candidate) amount = parse_amount(candidate)
except ValueError: except ValueError:
pass pass
break break
line_items.append({ line_items.append(
'description': desc, {
'amount': amount, "description": desc,
}) "amount": amount,
}
)
# Skip past amount + currency lines # Skip past amount + currency lines
i = j + 1 if amount is not None else i + 1 i = j + 1 if amount is not None else i + 1
while i < len(item_lines) and item_lines[i] == 'USD': while i < len(item_lines) and item_lines[i] == "USD":
i += 1 i += 1
# Key-value pairs after "More Detail" # Key-value pairs after "More Detail"
kv = {} kv = {}
detail_start = None detail_start = None
for i, line in enumerate(lines): for i, line in enumerate(lines):
if line == 'More Detail': if line == "More Detail":
detail_start = i + 1 detail_start = i + 1
break break
if detail_start: if detail_start:
kv_keys = ['PO ID', 'Department', 'Status', 'Last Opened', 'Order Date', kv_keys = [
'Acknowledged At', 'Revision Date', 'Payment Term', 'Req #'] "PO ID",
"Department",
"Status",
"Last Opened",
"Order Date",
"Acknowledged At",
"Revision Date",
"Payment Term",
"Req #",
]
for i in range(detail_start, len(lines)): for i in range(detail_start, len(lines)):
for k in kv_keys: for k in kv_keys:
if lines[i] == k and i + 1 < len(lines): if lines[i] == k and i + 1 < len(lines):
@ -90,63 +101,69 @@ def parse_po_email(body):
capturing = False capturing = False
skipping_blanks = False skipping_blanks = False
for i, line in enumerate(lines): for i, line in enumerate(lines):
if line == 'Shipping': if line == "Shipping":
shipping_count += 1 shipping_count += 1
if shipping_count == 2: if shipping_count == 2:
capturing = True capturing = True
skipping_blanks = True skipping_blanks = True
continue continue
elif capturing: elif capturing:
if skipping_blanks and line == '': if skipping_blanks and line == "":
continue continue
skipping_blanks = False skipping_blanks = False
if line in ('', 'Ship To Address') or line.startswith('---') or line.startswith('http'): if (
line in ("", "Ship To Address")
or line.startswith("---")
or line.startswith("http")
):
break break
if line in kv_keys or line == 'Supplier' or line == 'More Detail': if line in kv_keys or line == "Supplier" or line == "More Detail":
break break
ship_to_lines.append(line) ship_to_lines.append(line)
ship_to_raw = '\n'.join(ship_to_lines) if ship_to_lines else None ship_to_raw = "\n".join(ship_to_lines) if ship_to_lines else None
# Extract site code from ship-to # Extract site code from ship-to
site_code = None site_code = None
if ship_to_raw: if ship_to_raw:
# (CODE) pattern — e.g. "Amazon.com Services LLC (XSF2)" # (CODE) pattern — e.g. "Amazon.com Services LLC (XSF2)"
m = re.search(r'\(([A-Z0-9]{3,5})\)', ship_to_raw) m = re.search(r"\(([A-Z0-9]{3,5})\)", ship_to_raw)
if m: if m:
site_code = m.group(1) site_code = m.group(1)
# "LLC - CODE" — e.g. "Amazon.com Services LLC - WUT9" # "LLC - CODE" — e.g. "Amazon.com Services LLC - WUT9"
if not site_code: if not site_code:
m = re.search(r'(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b', ship_to_raw) m = re.search(r"(?:LLC|Inc)\s*-\s*([A-Z0-9]{3,5})\b", ship_to_raw)
if m: if m:
site_code = m.group(1) site_code = m.group(1)
# "ATTN: ... Station CODE" or "ATTN: ... DS - CODE" # "ATTN: ... Station CODE" or "ATTN: ... DS - CODE"
if not site_code: if not site_code:
m = re.search(r'ATTN:.*?(?:Station|DS)\s*[-–]?\s*([A-Z0-9]{3,5})\b', ship_to_raw) m = re.search(
r"ATTN:.*?(?:Station|DS)\s*[-–]?\s*([A-Z0-9]{3,5})\b", ship_to_raw
)
if m: if m:
site_code = m.group(1) site_code = m.group(1)
# "CODE - Amazon" at start of first line # "CODE - Amazon" at start of first line
if not site_code: if not site_code:
m = re.match(r'^([A-Z0-9]{3,5})\s*-\s*Amazon', ship_to_raw) m = re.match(r"^([A-Z0-9]{3,5})\s*-\s*Amazon", ship_to_raw)
if m: if m:
site_code = m.group(1) site_code = m.group(1)
if not site_code and line_items: if not site_code and line_items:
for li in line_items: for li in line_items:
m = re.match(r'^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b', li['description']) m = re.match(r"^([A-Z]{1,4}[0-9]{1,2}|[A-Z]{4,5})\b", li["description"])
if m: if m:
site_code = m.group(1) site_code = m.group(1)
break break
return { return {
'po_number': po_number, "po_number": po_number,
'site_code': site_code, "site_code": site_code,
'status': kv.get('Status'), "status": kv.get("Status"),
'order_date': kv.get('Order Date'), "order_date": kv.get("Order Date"),
'revision_date': kv.get('Revision Date'), "revision_date": kv.get("Revision Date"),
'payment_term': kv.get('Payment Term'), "payment_term": kv.get("Payment Term"),
'req_number': kv.get('Req #'), "req_number": kv.get("Req #"),
'ship_to_raw': ship_to_raw, "ship_to_raw": ship_to_raw,
'line_items': line_items, "line_items": line_items,
} }
@ -159,27 +176,29 @@ def process_mbox(mbox_path, target_pos=None, output_path=None):
for msg in mbox: for msg in mbox:
processed += 1 processed += 1
subject = msg.get('Subject', '') subject = msg.get("Subject", "")
if 'issued' not in subject.lower() or 'Purchase Order' not in subject: if "issued" not in subject.lower() or "Purchase Order" not in subject:
continue continue
if msg.is_multipart(): if msg.is_multipart():
body = None body = None
for part in msg.walk(): for part in msg.walk():
if part.get_content_type() == 'text/plain': if part.get_content_type() == "text/plain":
body = part.get_payload(decode=True).decode('utf-8', errors='replace') body = part.get_payload(decode=True).decode(
"utf-8", errors="replace"
)
break break
if not body: if not body:
continue continue
else: else:
body = msg.get_payload(decode=True).decode('utf-8', errors='replace') body = msg.get_payload(decode=True).decode("utf-8", errors="replace")
record = parse_po_email(body) record = parse_po_email(body)
if not record: if not record:
continue continue
po = record['po_number'] po = record["po_number"]
if target_pos and po not in target_pos: if target_pos and po not in target_pos:
continue continue
@ -190,25 +209,34 @@ def process_mbox(mbox_path, target_pos=None, output_path=None):
matched += 1 matched += 1
if matched % 500 == 0 and matched > 0: if matched % 500 == 0 and matched > 0:
print(f" Matched {matched} POs so far... ({processed} messages scanned)", file=sys.stderr) print(
f" Matched {matched} POs so far... ({processed} messages scanned)",
file=sys.stderr,
)
mbox.close() mbox.close()
result_list = list(results.values()) result_list = list(results.values())
print(f"Processed {processed} messages, matched {matched} unique POs", file=sys.stderr) print(
f"Processed {processed} messages, matched {matched} unique POs", file=sys.stderr
)
if output_path: if output_path:
with open(output_path, 'w') as f: with open(output_path, "w") as f:
json.dump(result_list, f, indent=2) json.dump(result_list, f, indent=2)
print(f"Saved to {output_path}", file=sys.stderr) print(f"Saved to {output_path}", file=sys.stderr)
return result_list return result_list
if __name__ == '__main__': if __name__ == "__main__":
mbox_path = sys.argv[1] if len(sys.argv) > 1 else '/tmp/coupa-emails/coupa-po-dump--info@seahavenind.com-wHxmHD.mbox' mbox_path = (
po_list_path = sys.argv[2] if len(sys.argv) > 2 else 'output/po-list.json' sys.argv[1]
output_path = sys.argv[3] if len(sys.argv) > 3 else 'output/email-parsed-pos.json' if len(sys.argv) > 1
else "/tmp/coupa-emails/coupa-po-dump--info@seahavenind.com-wHxmHD.mbox"
)
po_list_path = sys.argv[2] if len(sys.argv) > 2 else "output/po-list.json"
output_path = sys.argv[3] if len(sys.argv) > 3 else "output/email-parsed-pos.json"
target_pos = None target_pos = None
if os.path.exists(po_list_path): if os.path.exists(po_list_path):