(20241212) Day-end push.
This commit is contained in:
+134
-103
@@ -6,7 +6,7 @@
|
||||
|
||||
DATE:
|
||||
|
||||
Tuesday, 26th Nov., 2024
|
||||
Tuesday, 10th Dec., 2024
|
||||
|
||||
OBJECTIVE:
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
1. GitHub: https://github.com/SpamScope/mail-parser
|
||||
2. RFC 5322: https://datatracker.ietf.org/doc/html/rfc5322
|
||||
3. RFC 2045: https://datatracker.ietf.org/doc/html/rfc2045
|
||||
4. StackOverflow: https://stackoverflow.com/questions/17874360/python-how-to-parse-the-body-from-a-raw-email-given-that-raw-email-does-not
|
||||
|
||||
DOWNLOADS:
|
||||
|
||||
@@ -47,18 +48,26 @@ from utils_v2.string import regex
|
||||
from utils_v2.date_time import date_time
|
||||
|
||||
# To work with mails:
|
||||
import mailparser
|
||||
import email
|
||||
from email.message import Message
|
||||
from email.utils import parsedate_tz
|
||||
from email.utils import parseaddr
|
||||
|
||||
# To parse the HTML content in the mail:
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# To work with datatypes:
|
||||
from typing import Any, Dict
|
||||
from typing import Any, Dict, List, Literal
|
||||
|
||||
# To work with various encodings:
|
||||
import base64
|
||||
import quopri
|
||||
|
||||
# To work with date and time:
|
||||
import datetime
|
||||
import pytz
|
||||
import time
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
@@ -87,46 +96,116 @@ import quopri
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
def from_quoted_printable(text: str) -> str:
|
||||
def parse_addr(addr_header: str) -> List[Dict[str, str]]:
|
||||
|
||||
decoded_text = ""
|
||||
decoded_bytes = quopri.decodestring(text)
|
||||
for encoding in ["utf-8", "utf-16", "utf-32", "latin1"]:
|
||||
try: text = decoded_bytes.decode(encoding)
|
||||
except: text = ""
|
||||
if text.find("From") >= 0:
|
||||
decoded_text = text
|
||||
break
|
||||
return decoded_text
|
||||
# If the field is null, we return null:
|
||||
if addr_header is None: return []
|
||||
|
||||
# Create an empty variable that will hold the results:
|
||||
addrs = []
|
||||
|
||||
# Iterate through the addresses and parse them:
|
||||
for a in addr_header.split(","):
|
||||
n, e = parseaddr(a.strip())
|
||||
addrs.append({
|
||||
"name": n.strip() or e.strip(),
|
||||
"email": e.strip()
|
||||
})
|
||||
|
||||
# Done here:
|
||||
return addrs
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def find_in_raw_mail(
|
||||
raw_mail: str,
|
||||
text: str
|
||||
) -> int:
|
||||
def parse_date(date_header: str) -> datetime.datetime | None:
|
||||
|
||||
# We first treat it as un-encoded text:
|
||||
offset = raw_mail.find(text)
|
||||
if offset >= 0: return offset
|
||||
# Try to parse the date header:
|
||||
date_tuple = parsedate_tz(date_header)
|
||||
|
||||
# Then we try Base64 encoding:
|
||||
offset = raw_mail.find(base64.b64encode(text.encode("utf-8")).decode("utf-8"))
|
||||
if offset >= 0: return offset
|
||||
# If the date header was parsed successfully, we assemble
|
||||
# the parts to get an aware object in UTC timezone:
|
||||
if date_tuple:
|
||||
dt = datetime.datetime(*date_tuple[:6], tzinfo = pytz.FixedOffset(int(date_tuple[-1] / 60)))
|
||||
dt = date_time.to_timezone(dt, date_time.TIMEZONE_UTC)
|
||||
return dt
|
||||
|
||||
# then we try Quoted-Printable encoding:
|
||||
offset = from_quoted_printable(raw_mail).find(text)
|
||||
print("MAIL:")
|
||||
print(raw_mail)
|
||||
print("\n\n\n---\n\n\n")
|
||||
print("TEXT:")
|
||||
print(quopri.encodestring(text.encode("utf-8")).decode("utf-8"))
|
||||
if offset >= 0: return offset
|
||||
# In case of an invalid date header:
|
||||
else: return None
|
||||
|
||||
# Done here, even if nothing worked:
|
||||
return offset
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def decode_payload(
|
||||
raw_payload: str | bytes,
|
||||
content_main_type: str,
|
||||
content_charset: str | None,
|
||||
content_transfer_encoding: Literal[None, "base64", "quoted-printable"]
|
||||
) -> str | bytes:
|
||||
|
||||
# Start by assuming nothing needs to be done:
|
||||
payload = raw_payload
|
||||
|
||||
# We decode various kinds of parts:
|
||||
match content_transfer_encoding:
|
||||
|
||||
# This is just unencoded plaintext:
|
||||
case None:
|
||||
pass
|
||||
|
||||
# Typically see with attachments:
|
||||
case "base64":
|
||||
charset = content_charset or "utf-8"
|
||||
payload = raw_payload
|
||||
payload = base64.b64decode(payload)
|
||||
if content_main_type == "text": payload = payload.decode(charset)
|
||||
|
||||
# Typically seen with HTML parts:
|
||||
case "quoted-printable":
|
||||
charset = content_charset or "utf-8"
|
||||
payload = raw_payload.encode(charset)
|
||||
payload = quopri.decodestring(payload)
|
||||
if content_main_type == "text": payload = payload.decode(charset)
|
||||
|
||||
# Done here:
|
||||
return payload
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def parse_part(
|
||||
part: Message | List[Message]
|
||||
) -> Dict[str, Any]:
|
||||
|
||||
# Start by extracting basic details:
|
||||
part_json = {
|
||||
"boundary": part.get_boundary(),
|
||||
"contentType": part.get_content_type(),
|
||||
"contentMainType": part.get_content_maintype(),
|
||||
"contentSubType": part.get_content_subtype(),
|
||||
"contentCharset": part.get_content_charset(),
|
||||
"contentTransferEncoding": part.get("Content-Transfer-Encoding"),
|
||||
"contentDisposition": part.get_content_disposition(),
|
||||
"filename": part.get_filename(),
|
||||
"contentId": part.get("Content-ID")
|
||||
}
|
||||
|
||||
# Process the payload of this part:
|
||||
if part_json["contentMainType"] == "multipart":
|
||||
part_json["payload"] = [parse_part(sub_part) for sub_part in part.get_payload(decode = False)]
|
||||
else:
|
||||
part_json["payload"] = decode_payload(
|
||||
raw_payload = part.get_payload(decode = False),
|
||||
content_main_type = part_json["contentMainType"],
|
||||
content_charset = part_json["contentCharset"],
|
||||
content_transfer_encoding = part_json["contentTransferEncoding"]
|
||||
)
|
||||
|
||||
# Done here;
|
||||
return part_json
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------------------------
|
||||
@@ -145,78 +224,30 @@ def parse(raw_mail: str | bytes) -> Dict[str, Any]:
|
||||
"""
|
||||
|
||||
# Parse the raw format:
|
||||
if isinstance(raw_mail, str): parsed_mail = mailparser.parse_from_string(raw_mail)
|
||||
else: parsed_mail = mailparser.parse_from_bytes(raw_mail)
|
||||
if isinstance(raw_mail, str): parsed_mail = email.message_from_string(raw_mail)
|
||||
else: parsed_mail = email.message_from_bytes(raw_mail)
|
||||
|
||||
# print(parsed_mail.mail_json)
|
||||
# return
|
||||
|
||||
# Format the attachments:
|
||||
message_attachments = [
|
||||
{
|
||||
"filename": attachment["filename"],
|
||||
"type": attachment["mail_content_type"],
|
||||
"cid": regex.find_first(text = attachment["content-id"], pattern = r"(?<=<).*(?=>)"),
|
||||
"rawCid": attachment["content-id"],
|
||||
"contentDisposition": (cd := attachment["content-disposition"]),
|
||||
"isInline": True if cd.lower().find("inline") >= 0 else False,
|
||||
"charset": attachment["charset"],
|
||||
"contentTransferEncoding": attachment["content_transfer_encoding"],
|
||||
"payload": attachment["payload"]
|
||||
} for attachment in parsed_mail.attachments
|
||||
]
|
||||
|
||||
# print("PRINTING PARTS")
|
||||
# print("LIBRARY PARTS:", type(parsed_mail))
|
||||
|
||||
# # Figure out which entity (text and HTML) came in which sequence.
|
||||
# # The library doesn't give us any sequence info so we do some custom string processing here to figure out the order
|
||||
# # in which to render the contents of the page.
|
||||
# parts = [
|
||||
# # {
|
||||
# # "partNo": None,
|
||||
# # "offset": max(
|
||||
# # parsed_mail.message_as_string.find(t),
|
||||
# # parsed_mail.message_as_string.find(base64.b64encode(t.encode()).decode())
|
||||
# # ),
|
||||
# # "type": "text/plain",
|
||||
# # "data": t
|
||||
# # } for t in parsed_mail.text_plain
|
||||
# ]
|
||||
# parts = parts + [
|
||||
# {
|
||||
# "partNo": None,
|
||||
# "offset": find_in_raw_mail(raw_mail = parsed_mail.message_as_string, text = h),
|
||||
# "type": "text/html",
|
||||
# "data": h
|
||||
# } for h in parsed_mail.text_html
|
||||
# ]
|
||||
# parts = sorted(parts, key = lambda x: x["offset"])
|
||||
# for i, p in enumerate(parts): p["partNo"] = i
|
||||
|
||||
# Get the unformatted text from everything in the mail:
|
||||
unformatted_text = []
|
||||
for p in parsed_mail.text_html:
|
||||
html_parser = BeautifulSoup(p, "html.parser")
|
||||
unformatted_text.append(html_parser.get_text())
|
||||
|
||||
# Put everything together:
|
||||
return {
|
||||
"ts": date_time.to_timezone(parsed_mail.date, timezone = date_time.TIMEZONE_UTC),
|
||||
"headers": parsed_mail.headers,
|
||||
"from": [{"name": _[0] or _[1], "email": _[1]} for _ in parsed_mail.headers["From"]],
|
||||
"to": [{"name": _[0] or _[1], "email": _[1]} for _ in parsed_mail.headers["To"]],
|
||||
"cc": [{"name": _[0] or _[1], "email": _[1]} for _ in parsed_mail.headers.get("Cc", [])],
|
||||
"bcc": [{"name": _[0] or _[1], "email": _[1]} for _ in parsed_mail.headers.get("Bcc", [])],
|
||||
"subject": parsed_mail.headers["Subject"],
|
||||
"text": parsed_mail.text_plain,
|
||||
"html": parsed_mail.text_html,
|
||||
# "parts": parts,
|
||||
"unformattedText": "\n".join(unformatted_text),
|
||||
"attachments": message_attachments,
|
||||
"isInbox": None
|
||||
# Extract the most basic details:
|
||||
mail_json = {
|
||||
"ts": parse_date(parsed_mail["Date"]),
|
||||
"headers": {k: v for k, v in parsed_mail.items()},
|
||||
"from": parse_addr(parsed_mail["From"]),
|
||||
"to": parse_addr(parsed_mail["To"]),
|
||||
"cc": parse_addr(parsed_mail["Cc"]),
|
||||
"bcc": parse_addr(parsed_mail["Bcc"]),
|
||||
"subject": parsed_mail["Subject"],
|
||||
"payload": None
|
||||
}
|
||||
|
||||
# Iterate through each part of the mail for multipart mails:
|
||||
if parsed_mail.is_multipart(): mail_json["payload"] = parse_part(parsed_mail)
|
||||
|
||||
# When the mails are not multipart, just plaintext:
|
||||
else: mail_json["payload"] = parsed_mail.get_payload()
|
||||
|
||||
# Done here:
|
||||
return mail_json
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
@@ -229,7 +260,7 @@ if __name__ == "__main__":
|
||||
|
||||
from utils_v2.system import files
|
||||
|
||||
mail_string_raw = files.read_file(r"/home/developer/Downloads/raw_mail.txt")
|
||||
mail_string_raw = files.read_file(r"/home/developer/Downloads/raw_mail_test.txt")
|
||||
parse_results = parse(mail_string_raw)
|
||||
|
||||
print(json.to_string(parse_results, default = str))
|
||||
|
||||
Reference in New Issue
Block a user