(20241210) listing bug fix.
This commit is contained in:
@@ -133,8 +133,10 @@ class MailRetrieveModel(BaseModel):
|
||||
# Format the data:
|
||||
if mail_data:
|
||||
mail_data["mailId"] = str(mail_data.pop("_id"))
|
||||
mail_data["payload"]["ts"] = mail_data["payload"]["ts"].isoformat()
|
||||
mail_data["payload"]["readTs"] = mail_data["payload"]["readTs"].isoformat()
|
||||
# mail_data["payload"]["ts"] = mail_data["payload"]["ts"].isoformat()
|
||||
# mail_data["payload"]["readTs"] = mail_data["payload"]["readTs"].isoformat()
|
||||
mail_data["ts"] = mail_data["ts"].isoformat()
|
||||
mail_data["readTs"] = mail_data["readTs"].isoformat()
|
||||
|
||||
# Done here:
|
||||
return mail_data
|
||||
@@ -168,6 +170,8 @@ class MailRetrieveModel(BaseModel):
|
||||
"_id": True,
|
||||
"serviceType": True,
|
||||
"client": True,
|
||||
"ts": True,
|
||||
"readTs": True,
|
||||
"payload.ts": True,
|
||||
"payload.readTs": True,
|
||||
"payload.from": True,
|
||||
@@ -185,8 +189,10 @@ class MailRetrieveModel(BaseModel):
|
||||
if mails_list:
|
||||
for mail_data in mails_list:
|
||||
mail_data["mailId"] = str(mail_data.pop("_id"))
|
||||
mail_data["payload"]["ts"] = mail_data["payload"]["ts"].isoformat()
|
||||
mail_data["payload"]["readTs"] = mail_data["payload"]["readTs"].isoformat()
|
||||
# mail_data["payload"]["ts"] = mail_data["payload"]["ts"].isoformat()
|
||||
# mail_data["payload"]["readTs"] = mail_data["payload"]["readTs"].isoformat()
|
||||
mail_data["ts"] = mail_data["ts"].isoformat()
|
||||
mail_data["readTs"] = mail_data["readTs"].isoformat()
|
||||
|
||||
# Done here:
|
||||
return mails_list
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
"""
|
||||
|
||||
AUTHOR:
|
||||
|
||||
Khushal P Soonderji
|
||||
|
||||
DATE:
|
||||
|
||||
Tuesday, 10th Dec., 2024
|
||||
|
||||
OBJECTIVE:
|
||||
|
||||
To parse raw mail bodies and give a structure that is suitable for storing in No-SQL databases like MongoDB. The
|
||||
raw mail's text is expected to be compliant with standard defined in RFC 5322, RFC 2045, and maybe a few more.
|
||||
|
||||
REFERENCES:
|
||||
|
||||
1. GitHub: https://github.com/SpamScope/mail-parser
|
||||
2. RFC 5322: https://datatracker.ietf.org/doc/html/rfc5322
|
||||
3. RFC 2045: https://datatracker.ietf.org/doc/html/rfc2045
|
||||
4. StackOverflow: https://stackoverflow.com/questions/17874360/python-how-to-parse-the-body-from-a-raw-email-given-that-raw-email-does-not
|
||||
|
||||
DOWNLOADS:
|
||||
|
||||
N/A
|
||||
|
||||
"""
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** IMPORT ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
# To make sibling directories accessible for imports:
|
||||
import sys
|
||||
sys.path.append(".")
|
||||
sys.path.append("..")
|
||||
|
||||
# System-level activities:
|
||||
import io
|
||||
|
||||
# My utils:
|
||||
from utils_v2.string import json
|
||||
from utils_v2.string import regex
|
||||
from utils_v2.date_time import date_time
|
||||
|
||||
# To work with mails:
|
||||
import email
|
||||
|
||||
# To parse the HTML content in the mail:
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# To work with datatypes:
|
||||
from typing import Any, Dict
|
||||
|
||||
# To work with various encodings:
|
||||
import base64
|
||||
import quopri
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** MACROS / ONE-TIME INIT ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
# --- Nothing Yet
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** VARIABLES ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
# --- Nothing Yet
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** FUNCTIONS ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
def parse(raw_mail: str | bytes) -> Dict[str, Any]:
|
||||
|
||||
"""
|
||||
To parse the raw mail text to a usable JSON that can even be stored on a No-SQL database like MongoDB.
|
||||
DOCUMENTATION:
|
||||
1. GitHub: https://github.com/SpamScope/mail-parser
|
||||
2. RFC 5322: https://datatracker.ietf.org/doc/html/rfc5322
|
||||
3. RFC 2045: https://datatracker.ietf.org/doc/html/rfc2045
|
||||
:param raw_mail: The raw mail body that adheres to RFC 5322 and RFC 2045 (among others).
|
||||
:return: The parsed JSON format (dict) of the mail.
|
||||
"""
|
||||
|
||||
# Parse the raw format:
|
||||
if isinstance(raw_mail, str): parsed_mail = email.message_from_string(raw_mail)
|
||||
else: parsed_mail = email.message_from_bytes(raw_mail)
|
||||
|
||||
# Make variables:
|
||||
parts = []
|
||||
|
||||
# Iterate through each part of the mail for multipart mails:
|
||||
if parsed_mail.is_multipart():
|
||||
for part in parsed_mail.walk():
|
||||
print("MULTIPART PART:", part)
|
||||
|
||||
# When the mails are not multipart, just plaintext:
|
||||
else: print("PLAINTEXT PART:", parsed_mail.get_payload())
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** MAIN PROGRAM ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
from utils_v2.system import files
|
||||
|
||||
mail_string_raw = files.read_file(r"/home/developer/Downloads/raw_mail.txt")
|
||||
parse_results = parse(mail_string_raw)
|
||||
|
||||
print(json.to_string(parse_results, default = str))
|
||||
Reference in New Issue
Block a user