"""IMAP mail ingestion. Auth is a plain password via EMAIL_PASSWORD, not OAuth2/XOAUTH2: for Gmail that means an App Password (enable 2FA, then generate one at myaccount.google.com -> Security -> App passwords); other providers have an equivalent app-specific password. OAuth2 would need a browser consent round-trip plus refresh-token storage, which is far too heavy for a single-user cron-style job — App Password is the documented, supported path for exactly this case. """ import email import email.utils import hashlib import logging import os import re from datetime import datetime, timedelta, timezone from email.header import decode_header, make_header from pathlib import Path LOG = logging.getLogger(__name__) MAX_BODY_CHARS = 2000 # Where document attachments are spooled for agenda.py to read. Inside the digest's # own /data volume, alongside the Telegram session and the archive database. DEFAULT_ATTACHMENT_DIR = "/data/attachments" DOCUMENT_SUFFIXES = (".pdf", ".doc", ".docx", ".odt", ".rtf", ".txt", ".md") DOCUMENT_TYPES = { "application/pdf", "application/msword", "application/vnd.openxmlformats-officedocument.wordprocessingml.document", "application/vnd.oasis.opendocument.text", "application/rtf", "text/plain", } def _decode(value): if not value: return "" try: return str(make_header(decode_header(value))) except Exception: return str(value) def _body_text(message): if message.is_multipart(): for part in message.walk(): if part.get_content_type() == "text/plain" and "attachment" not in str( part.get("Content-Disposition", "") ): payload = part.get_payload(decode=True) if payload: return payload.decode(part.get_content_charset() or "utf-8", "replace") return "" payload = message.get_payload(decode=True) if not payload: return "" return payload.decode(message.get_content_charset() or "utf-8", "replace") def _safe_name(filename, fallback): """A filename that cannot escape the spool directory. Takes the basename, keeps only characters that are unambiguously safe, and prefixes a short hash of the original so two "Tagesordnung.pdf"s from different senders can coexist.""" base = os.path.basename(filename or "").strip() or fallback cleaned = re.sub(r"[^A-Za-z0-9._-]+", "_", base)[:80].lstrip(".") or fallback digest = hashlib.sha256((filename or fallback).encode("utf-8")).hexdigest()[:8] return f"{digest}-{cleaned}" def _attachments(message, uid): """Attachment metadata, plus the file itself for documents. Documents are written to DIGEST_ATTACHMENT_DIR so `agenda.py` can read a meeting agenda's text out of them. Everything else (images, calendar invites, the usual signature logos) is recorded by name and type only — this is a digest, not a mail archive, and there is no reason to spool a 4 MB photo to disk. This writes to the digest's own data volume, exactly as `context.json` already does. It is not a write against the mailbox: the message is fetched read-only, the folder stays unread, and nothing is flagged, moved or deleted. """ if not message.is_multipart(): return [] spool = Path(os.environ.get("DIGEST_ATTACHMENT_DIR", DEFAULT_ATTACHMENT_DIR)) max_bytes = int(os.environ.get("EMAIL_MAX_ATTACHMENT_MB", "10")) * 1024 * 1024 save_documents = os.environ.get("EMAIL_SAVE_DOCUMENTS", "true").strip().lower() == "true" found = [] for index, part in enumerate(message.walk()): filename = part.get_filename() disposition = str(part.get("Content-Disposition", "")) if not filename and "attachment" not in disposition: continue name = _decode(filename) if filename else "" content_type = part.get_content_type() entry = {"filename": name, "content_type": content_type, "path": None} is_document = name.lower().endswith(DOCUMENT_SUFFIXES) or content_type in DOCUMENT_TYPES if save_documents and is_document: try: payload = part.get_payload(decode=True) or b"" if len(payload) > max_bytes: LOG.info("email: attachment %r is %d bytes, not spooling it", name, len(payload)) else: spool.mkdir(parents=True, exist_ok=True) target = spool / _safe_name(name, f"uid{uid}-part{index}") target.write_bytes(payload) entry["path"] = str(target) entry["bytes"] = len(payload) except Exception: # A document that cannot be spooled is still worth reporting by name. LOG.warning("email: could not save attachment %r", name, exc_info=True) found.append(entry) return found def fetch(lookback_hours): host = os.environ.get("EMAIL_IMAP_HOST", "").strip() user = os.environ.get("EMAIL_USERNAME", "").strip() password = os.environ.get("EMAIL_PASSWORD", "") folder = os.environ.get("EMAIL_FOLDER", "INBOX") port = int(os.environ.get("EMAIL_IMAP_PORT", "993")) lookback_hours = float(os.environ.get("EMAIL_LOOKBACK_HOURS", lookback_hours)) if not (host and user and password): LOG.warning("email: EMAIL_IMAP_HOST/EMAIL_USERNAME/EMAIL_PASSWORD not all set, skipping") return [] since = datetime.now(timezone.utc) - timedelta(hours=lookback_hours) messages = [] try: # Imported here, not at module scope, so a missing/broken optional # dependency degrades this one source instead of the whole run. from imapclient import IMAPClient with IMAPClient(host, port=port, ssl=True) as client: client.login(user, password) # readonly=True is the whole point: it stops the server from setting # \Seen on anything we FETCH. digest-engine never mutates the mailbox. client.select_folder(folder, readonly=True) uids = client.search(["SINCE", since.date()]) if not uids: return [] for uid, data in client.fetch(uids, ["RFC822"]).items(): raw = data.get(b"RFC822") if not raw: continue try: parsed = email.message_from_bytes(raw) sent_at = email.utils.parsedate_to_datetime(parsed.get("Date")) if sent_at is not None and sent_at.tzinfo is None: sent_at = sent_at.replace(tzinfo=timezone.utc) if sent_at is not None and sent_at < since: continue messages.append( { "source": "email", "uid": int(uid), "from": _decode(parsed.get("From")), "to": _decode(parsed.get("To")), "subject": _decode(parsed.get("Subject")), "timestamp": sent_at.isoformat() if sent_at else None, "body": _body_text(parsed).strip()[:MAX_BODY_CHARS], "attachments": _attachments(parsed, uid), } ) except Exception: LOG.warning("email: could not parse message uid=%s, skipping", uid, exc_info=True) except Exception: LOG.warning("email: ingestion failed, returning nothing", exc_info=True) return [] LOG.info("email: %d message(s) in the last %sh", len(messages), lookback_hours) return messages