mirror of
https://github.com/Sea-Haven-Industries/procurement-ingest.git
synced 2026-10-01 18:13:13 +00:00
35 lines
905 B
Python
35 lines
905 B
Python
|
|
import email
|
||
|
|
from email import policy
|
||
|
|
|
||
|
|
|
||
|
|
def parse_raw_email(raw_bytes: bytes) -> dict:
|
||
|
|
"""Parse a raw email into subject, sender, body text."""
|
||
|
|
msg = email.message_from_bytes(raw_bytes, policy=policy.default)
|
||
|
|
|
||
|
|
subject = msg.get("Subject", "")
|
||
|
|
sender = msg.get("From", "")
|
||
|
|
to = msg.get("To", "")
|
||
|
|
cc = msg.get("Cc", "")
|
||
|
|
date = msg.get("Date", "")
|
||
|
|
|
||
|
|
body = ""
|
||
|
|
if msg.is_multipart():
|
||
|
|
for part in msg.walk():
|
||
|
|
content_type = part.get_content_type()
|
||
|
|
if content_type == "text/plain":
|
||
|
|
body = part.get_content()
|
||
|
|
break
|
||
|
|
elif content_type == "text/html" and not body:
|
||
|
|
body = part.get_content()
|
||
|
|
else:
|
||
|
|
body = msg.get_content()
|
||
|
|
|
||
|
|
return {
|
||
|
|
"subject": subject,
|
||
|
|
"sender": sender,
|
||
|
|
"to": to,
|
||
|
|
"cc": cc,
|
||
|
|
"date": date,
|
||
|
|
"body": body,
|
||
|
|
}
|