Files
mail_summary/backend/mail_parser.py
T
2026-07-19 23:04:16 +09:00

343 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Zip(.eml) extraction, parsing, cleaning, and date filtering."""
from __future__ import annotations
import email
import email.policy
import re
import zipfile
from collections.abc import Callable
from dataclasses import dataclass, field
from datetime import date, datetime, timezone
from email.header import decode_header, make_header
from email.message import EmailMessage
from email.utils import parsedate_to_datetime
from html.parser import HTMLParser
from pathlib import Path
from zoneinfo import ZoneInfo
KST = ZoneInfo("Asia/Seoul")
_SUBJECT_PREFIX_RE = re.compile(
r"^\s*((re|fw|fwd|답장|전달|회신)\s*[:]|\[?\s*(re|fw|fwd)\s*\]?\s*:)\s*",
re.IGNORECASE,
)
_QUOTE_MARKERS = (
"-----original message-----",
"----- 원본 메시지 -----",
"-----원본 메시지-----",
"________________________________",
"begin forwarded message",
"보낸 사람:",
"보내는 사람:",
"보내는사람:",
"보내는사람 :",
"from:",
)
_SIG_MARKERS = (
"\n-- \n",
"\n--\n",
"\n감사합니다.\n",
"\nthanks,\n",
"\nbest regards,\n",
"\ncustomer centric agility\n",
"\ntel :",
"\ntel:",
)
@dataclass
class ParsedMail:
mailbox: str
message_id: str
thread_key: str
subject: str
from_addr: str
to_addrs: str
cc_addrs: str
mailed_at: datetime | None
attachment_names: list[str] = field(default_factory=list)
body_text: str = ""
source_path: str = ""
class _HTMLTextExtractor(HTMLParser):
def __init__(self) -> None:
super().__init__()
self._parts: list[str] = []
self._skip = False
def handle_starttag(self, tag: str, attrs) -> None: # noqa: ANN001
if tag in {"script", "style"}:
self._skip = True
elif tag in {"br", "p", "div", "tr", "li"}:
self._parts.append("\n")
def handle_endtag(self, tag: str) -> None:
if tag in {"script", "style"}:
self._skip = False
elif tag in {"p", "div", "tr", "li"}:
self._parts.append("\n")
def handle_data(self, data: str) -> None:
if not self._skip:
self._parts.append(data)
def text(self) -> str:
return "".join(self._parts)
def decode_mime_header(value: str | None) -> str:
if not value:
return ""
try:
return str(make_header(decode_header(value))).strip()
except Exception: # noqa: BLE001
return value.strip()
def normalize_thread_key(subject: str) -> str:
text = decode_mime_header(subject)
prev = None
while prev != text:
prev = text
text = _SUBJECT_PREFIX_RE.sub("", text).strip()
text = re.sub(r"\s+", " ", text).strip().lower()
return text or "(no-subject)"
def html_to_text(html: str) -> str:
parser = _HTMLTextExtractor()
try:
parser.feed(html)
parser.close()
except Exception: # noqa: BLE001
return re.sub(r"<[^>]+>", " ", html)
return parser.text()
def clean_body(text: str) -> str:
if not text:
return ""
normalized = text.replace("\r\n", "\n").replace("\r", "\n")
lower = normalized.lower()
cut = len(normalized)
for marker in _QUOTE_MARKERS:
idx = lower.find(marker)
if idx != -1:
cut = min(cut, idx)
body = normalized[:cut]
for marker in _SIG_MARKERS:
idx = body.lower().find(marker.lower())
if idx != -1 and idx > 40:
body = body[:idx]
break
lines: list[str] = []
for line in body.split("\n"):
stripped = line.strip()
if stripped.startswith(">"):
continue
if re.match(
r"^(from|sent|to|cc|subject|보내는\s*사람|보내는사람)\s*:",
stripped,
re.IGNORECASE,
):
# Likely start of an inline quoted header block.
break
if re.match(r"^customer centric agility$", stripped, re.IGNORECASE):
break
if re.match(r"^tel\s*:", stripped, re.IGNORECASE):
break
lines.append(line.rstrip())
cleaned = "\n".join(lines)
cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).strip()
return cleaned
def _addresses(msg: EmailMessage, header: str) -> str:
raw = msg.get_all(header, [])
if not raw:
return ""
values = [decode_mime_header(str(item)) for item in raw]
return ", ".join(v for v in values if v)
def _parse_date(msg: EmailMessage) -> datetime | None:
raw = msg.get("Date")
if not raw:
return None
try:
dt = parsedate_to_datetime(raw)
except Exception: # noqa: BLE001
return None
if dt.tzinfo is None:
dt = dt.replace(tzinfo=timezone.utc)
return dt.astimezone(KST)
def _extract_body(msg: EmailMessage) -> str:
text_part = ""
html_part = ""
if msg.is_multipart():
for part in msg.walk():
ctype = part.get_content_type()
disp = str(part.get("Content-Disposition") or "")
if "attachment" in disp.lower():
continue
try:
payload = part.get_content()
except Exception: # noqa: BLE001
continue
if not isinstance(payload, str):
continue
if ctype == "text/plain" and not text_part:
text_part = payload
elif ctype == "text/html" and not html_part:
html_part = payload
else:
try:
payload = msg.get_content()
except Exception: # noqa: BLE001
payload = ""
if isinstance(payload, str):
if msg.get_content_type() == "text/html":
html_part = payload
else:
text_part = payload
raw = text_part if text_part.strip() else html_to_text(html_part)
return clean_body(raw)
def _attachment_names(msg: EmailMessage) -> list[str]:
names: list[str] = []
for part in msg.walk():
filename = part.get_filename()
if filename:
names.append(decode_mime_header(filename))
return names
def parse_eml_bytes(data: bytes, mailbox: str, source_name: str) -> ParsedMail | None:
try:
msg = email.message_from_bytes(data, policy=email.policy.default)
except Exception: # noqa: BLE001
return None
subject = decode_mime_header(msg.get("Subject"))
message_id = decode_mime_header(msg.get("Message-ID")) or Path(source_name).name
return ParsedMail(
mailbox=mailbox,
message_id=message_id,
thread_key=normalize_thread_key(subject),
subject=subject,
from_addr=_addresses(msg, "From"),
to_addrs=_addresses(msg, "To"),
cc_addrs=_addresses(msg, "Cc"),
mailed_at=_parse_date(msg),
attachment_names=_attachment_names(msg),
body_text=_extract_body(msg),
source_path=source_name,
)
def parse_eml_file(path: Path, mailbox: str) -> ParsedMail | None:
try:
return parse_eml_bytes(path.read_bytes(), mailbox, str(path))
except Exception: # noqa: BLE001
return None
def find_zip(job_dir: Path, prefix: str) -> Path | None:
matches = sorted(job_dir.glob(f"{prefix}_*.zip"))
return matches[0] if matches else None
def iter_eml_members(zip_path: Path) -> list[zipfile.ZipInfo]:
members: list[zipfile.ZipInfo] = []
with zipfile.ZipFile(zip_path, "r") as zf:
for info in zf.infolist():
if info.is_dir():
continue
name = info.filename.replace("\\", "/")
if name.startswith("/") or ".." in Path(name).parts:
continue
if not name.lower().endswith(".eml"):
continue
members.append(info)
return members
def load_mailbox_from_zip(
zip_path: Path,
mailbox: str,
*,
date_from: date | None = None,
date_to: date | None = None,
on_progress: Callable[[int, int, int], None] | None = None,
) -> tuple[list[ParsedMail], int]:
"""Stream-parse .eml entries inside zip. Optionally keep only mails in date range.
Returns (mails, total_eml_count). Avoids extracting multi-GB archives to disk.
on_progress(scanned, total, matched) is called periodically.
"""
members = iter_eml_members(zip_path)
total = len(members)
mails: list[ParsedMail] = []
scanned = 0
if on_progress:
on_progress(0, total, 0)
with zipfile.ZipFile(zip_path, "r") as zf:
for info in members:
scanned += 1
name = info.filename.replace("\\", "/")
try:
with zf.open(info) as src:
data = src.read()
except Exception: # noqa: BLE001
data = b""
if data:
parsed = parse_eml_bytes(data, mailbox, name)
if parsed:
keep = True
if date_from and date_to:
if parsed.mailed_at is None:
keep = False
else:
day = parsed.mailed_at.astimezone(KST).date()
keep = date_from <= day <= date_to
if keep:
mails.append(parsed)
if on_progress and (scanned == total or scanned % 25 == 0):
on_progress(scanned, total, len(mails))
mails.sort(key=lambda m: m.mailed_at or datetime.min.replace(tzinfo=KST))
if on_progress:
on_progress(total, total, len(mails))
return mails, total
def filter_by_date(mails: list[ParsedMail], date_from: date, date_to: date) -> list[ParsedMail]:
filtered: list[ParsedMail] = []
for mail in mails:
if mail.mailed_at is None:
continue
local_day = mail.mailed_at.astimezone(KST).date()
if date_from <= local_day <= date_to:
filtered.append(mail)
filtered.sort(key=lambda m: m.mailed_at or datetime.min.replace(tzinfo=KST))
return filtered
def group_by_thread(mails: list[ParsedMail]) -> dict[str, list[ParsedMail]]:
groups: dict[str, list[ParsedMail]] = {}
for mail in mails:
groups.setdefault(mail.thread_key, []).append(mail)
for key in groups:
groups[key].sort(key=lambda m: m.mailed_at or datetime.min.replace(tzinfo=KST))
return groups