Coverage for paperless/parsers/mail.py: 19%
286 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1"""
2Built-in mail document parser.
4Handles message/rfc822 (EML) MIME type by:
5- Parsing the email using imap_tools
6- Generating a PDF via Gotenberg (for display and archive)
7- Extracting text via Tika for HTML content
8- Extracting metadata from email headers
10The parser always produces a PDF because EML files cannot be rendered
11natively in a browser (requires_pdf_rendition=True).
12"""
14from __future__ import annotations
16import logging
17import re
18import shutil
19import tempfile
20from html import escape
21from pathlib import Path
22from typing import TYPE_CHECKING
23from typing import Self
25from django.conf import settings
26from django.utils import timezone
27from django.utils.timezone import is_naive
28from django.utils.timezone import make_aware
29from gotenberg_client import GotenbergClient
30from gotenberg_client.constants import A4
31from gotenberg_client.options import Measurement
32from gotenberg_client.options import MeasurementUnitType
33from gotenberg_client.options import PageMarginsType
34from gotenberg_client.options import PdfAFormat
35from humanize import naturalsize
36from imap_tools import MailAttachment
37from imap_tools import MailMessage
38from tika_client import TikaClient
39from turbohtml.clean import Linkify
40from turbohtml.clean import linkify
41from turbohtml.migration.bleach import clean
43from documents.parsers import ParseError
44from documents.parsers import make_thumbnail_from_pdf
45from paperless.models import OutputTypeChoices
46from paperless.version import __full_version_str__
47from paperless_mail.models import MailRule
49if TYPE_CHECKING: 49 ↛ 50line 49 didn't jump to line 50 because the condition on line 49 was never true
50 import datetime
51 from types import TracebackType
53 from paperless.parsers import MetadataEntry
54 from paperless.parsers import ParserContext
56logger = logging.getLogger("paperless.parsing.mail")
58_SUPPORTED_MIME_TYPES: dict[str, str] = {
59 "message/rfc822": ".eml",
60}
63class MailDocumentParser:
64 """Parse .eml email files for Paperless-ngx.
66 Uses imap_tools to parse .eml files, generates a PDF using Gotenberg,
67 and sends the HTML part to a Tika server for text extraction. Because
68 EML files cannot be rendered natively in a browser, the parser always
69 produces a PDF rendition (requires_pdf_rendition=True).
71 Pass a ``ParserContext`` to ``configure()`` before ``parse()`` to
72 apply mail-rule-specific PDF layout options:
74 parser.configure(ParserContext(mailrule_id=rule.pk))
75 parser.parse(path, mime_type)
77 Class attributes
78 ----------------
79 name : str
80 Human-readable parser name.
81 version : str
82 Semantic version string, kept in sync with Paperless-ngx releases.
83 author : str
84 Maintainer name.
85 url : str
86 Issue tracker / source URL.
87 """
89 name: str = "Paperless-ngx Mail Parser"
90 version: str = __full_version_str__
91 author: str = "Paperless-ngx Contributors"
92 url: str = "https://github.com/paperless-ngx/paperless-ngx"
94 # ------------------------------------------------------------------
95 # Class methods
96 # ------------------------------------------------------------------
98 @classmethod
99 def supported_mime_types(cls) -> dict[str, str]:
100 """Return the MIME types this parser handles.
102 Returns
103 -------
104 dict[str, str]
105 Mapping of MIME type to preferred file extension.
106 """
107 return _SUPPORTED_MIME_TYPES
109 @classmethod
110 def score(
111 cls,
112 mime_type: str,
113 filename: str,
114 path: Path | None = None,
115 ) -> int | None:
116 """Return the priority score for handling this file.
118 Parameters
119 ----------
120 mime_type:
121 Detected MIME type of the file.
122 filename:
123 Original filename including extension.
124 path:
125 Optional filesystem path. Not inspected by this parser.
127 Returns
128 -------
129 int | None
130 10 if the MIME type is supported, otherwise None.
131 """
132 if mime_type in _SUPPORTED_MIME_TYPES:
133 return 10
134 return None
136 # ------------------------------------------------------------------
137 # Properties
138 # ------------------------------------------------------------------
140 @property
141 def can_produce_archive(self) -> bool:
142 """Whether this parser can produce a searchable PDF archive copy.
144 Returns
145 -------
146 bool
147 Always False — the mail parser produces a display PDF
148 (requires_pdf_rendition=True), not an optional OCR archive.
149 """
150 return False
152 @property
153 def requires_pdf_rendition(self) -> bool:
154 """Whether the parser must produce a PDF for the frontend to display.
156 Returns
157 -------
158 bool
159 Always True — EML files cannot be rendered natively in a browser,
160 so a PDF conversion is always required for display.
161 """
162 return True
164 # ------------------------------------------------------------------
165 # Lifecycle
166 # ------------------------------------------------------------------
168 def __init__(self, logging_group: object = None) -> None:
169 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True)
170 self._tempdir = Path(
171 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR),
172 )
173 self._text: str | None = None
174 self._date: datetime.datetime | None = None
175 self._archive_path: Path | None = None
176 self._mailrule_id: int | None = None
178 def __enter__(self) -> Self:
179 return self
181 def __exit__(
182 self,
183 exc_type: type[BaseException] | None,
184 exc_val: BaseException | None,
185 exc_tb: TracebackType | None,
186 ) -> None:
187 logger.debug("Cleaning up temporary directory %s", self._tempdir)
188 shutil.rmtree(self._tempdir, ignore_errors=True)
190 # ------------------------------------------------------------------
191 # Core parsing interface
192 # ------------------------------------------------------------------
194 def configure(self, context: ParserContext) -> None:
195 self._mailrule_id = context.mailrule_id
197 def parse(
198 self,
199 document_path: Path,
200 mime_type: str,
201 *,
202 produce_archive: bool = True,
203 ) -> None:
204 """Parse the given .eml into formatted text and a PDF archive.
206 Call ``configure(ParserContext(mailrule_id=...))`` before this method
207 to apply mail-rule-specific PDF layout options. The ``produce_archive``
208 flag is accepted for protocol compatibility but is always honoured —
209 the mail parser always produces a PDF since EML files cannot be
210 displayed natively.
212 Parameters
213 ----------
214 document_path:
215 Absolute path to the .eml file.
216 mime_type:
217 Detected MIME type of the document (should be "message/rfc822").
218 produce_archive:
219 Accepted for protocol compatibility. The PDF rendition is always
220 produced since EML files cannot be displayed natively in a browser.
222 Raises
223 ------
224 documents.parsers.ParseError
225 If the file cannot be parsed or PDF generation fails.
226 """
228 def strip_text(text: str) -> str:
229 """Reduces the spacing of the given text string."""
230 text = re.sub(r"\s+", " ", text)
231 text = re.sub(r"(\n *)+", "\n", text)
232 return text.strip()
234 def build_formatted_text(mail_message: MailMessage) -> str:
235 """Constructs a formatted string based on the given email."""
236 fmt_text = f"Subject: {mail_message.subject}\n\n"
237 fmt_text += f"From: {mail_message.from_values.full if mail_message.from_values else ''}\n\n"
238 to_list = [address.full for address in mail_message.to_values]
239 fmt_text += f"To: {', '.join(to_list)}\n\n"
240 if mail_message.cc_values:
241 fmt_text += (
242 f"CC: {', '.join(address.full for address in mail.cc_values)}\n\n"
243 )
244 if mail_message.bcc_values:
245 fmt_text += (
246 f"BCC: {', '.join(address.full for address in mail.bcc_values)}\n\n"
247 )
248 if mail_message.attachments:
249 att = []
250 for a in mail.attachments:
251 attachment_size = naturalsize(a.size, binary=True, format="%.2f")
252 att.append(
253 f"{a.filename} ({attachment_size})",
254 )
255 fmt_text += f"Attachments: {', '.join(att)}\n\n"
257 if mail.html:
258 fmt_text += "HTML content: " + strip_text(self.tika_parse(mail.html))
260 fmt_text += f"\n\n{strip_text(mail.text)}"
262 return fmt_text
264 logger.debug("Parsing file %s into an email", document_path.name)
265 mail = self.parse_file_to_message(document_path)
267 logger.debug("Building formatted text from email")
268 self._text = build_formatted_text(mail)
270 self._date = mail.date
272 logger.debug("Creating a PDF from the email")
273 if self._mailrule_id:
274 rule = MailRule.objects.get(pk=self._mailrule_id)
275 self._archive_path = self.generate_pdf(
276 mail,
277 MailRule.PdfLayout(rule.pdf_layout),
278 )
279 else:
280 self._archive_path = self.generate_pdf(mail)
282 # ------------------------------------------------------------------
283 # Result accessors
284 # ------------------------------------------------------------------
286 def get_text(self) -> str:
287 """Return the plain-text content extracted during parse.
289 Returns
290 -------
291 str
292 Extracted text, or an empty string if no text could be found.
293 """
294 return self._text or ""
296 def get_date(self) -> datetime.datetime | None:
297 """Return the document date detected during parse.
299 Returns
300 -------
301 datetime.datetime | None
302 Date from the email headers, or None if not detected.
303 """
304 return self._date
306 def get_archive_path(self) -> Path | None:
307 """Return the path to the generated archive PDF, or None.
309 Returns
310 -------
311 Path | None
312 Path to the PDF produced by Gotenberg, or None if parse has not
313 been called yet.
314 """
315 return self._archive_path
317 # ------------------------------------------------------------------
318 # Thumbnail and metadata
319 # ------------------------------------------------------------------
321 def get_thumbnail(
322 self,
323 document_path: Path,
324 mime_type: str,
325 file_name: str | None = None,
326 ) -> Path:
327 """Generate a thumbnail from the PDF rendition of the email.
329 Converts the document to PDF first if not already done.
331 Parameters
332 ----------
333 document_path:
334 Absolute path to the source document.
335 mime_type:
336 Detected MIME type of the document.
337 file_name:
338 Kept for backward compatibility; not used.
340 Returns
341 -------
342 Path
343 Path to the generated WebP thumbnail inside the temporary directory.
344 """
345 if not self._archive_path:
346 self._archive_path = self.generate_pdf(
347 self.parse_file_to_message(document_path),
348 )
350 return make_thumbnail_from_pdf(
351 self._archive_path,
352 self._tempdir,
353 )
355 def get_page_count(
356 self,
357 document_path: Path,
358 mime_type: str,
359 ) -> int | None:
360 """Return the number of pages in the document.
362 Counts pages in the archive PDF produced by a preceding parse()
363 call. Returns ``None`` if parse() has not been called yet or if
364 no archive was produced.
366 Returns
367 -------
368 int | None
369 Page count of the archive PDF, or ``None``.
370 """
371 if self._archive_path is not None:
372 from paperless.parsers.utils import get_page_count_for_pdf
374 return get_page_count_for_pdf(self._archive_path, log=logger)
375 return None
377 def extract_metadata(
378 self,
379 document_path: Path,
380 mime_type: str,
381 ) -> list[MetadataEntry]:
382 """Extract metadata from the email headers.
384 Returns email headers as metadata entries with prefix "header",
385 plus summary entries for attachments and date.
387 Returns
388 -------
389 list[MetadataEntry]
390 Sorted list of metadata entries, or ``[]`` on parse failure.
391 """
392 result: list[MetadataEntry] = []
394 try:
395 mail = self.parse_file_to_message(document_path)
396 except ParseError as e:
397 logger.warning(
398 "Error while fetching document metadata for %s: %s",
399 document_path,
400 e,
401 )
402 return result
404 for key, header_values in mail.headers.items():
405 value = ", ".join(header_values)
406 try:
407 value.encode("utf-8")
408 except UnicodeEncodeError as e: # pragma: no cover
409 logger.debug("Skipping header %s: %s", key, e)
410 continue
412 result.append(
413 {
414 "namespace": "",
415 "prefix": "header",
416 "key": key,
417 "value": value,
418 },
419 )
421 result.append(
422 {
423 "namespace": "",
424 "prefix": "",
425 "key": "attachments",
426 "value": ", ".join(
427 f"{attachment.filename}"
428 f"({naturalsize(attachment.size, binary=True, format='%.2f')})"
429 for attachment in mail.attachments
430 ),
431 },
432 )
434 result.append(
435 {
436 "namespace": "",
437 "prefix": "",
438 "key": "date",
439 "value": mail.date.strftime("%Y-%m-%d %H:%M:%S %Z"),
440 },
441 )
443 result.sort(key=lambda item: (item["prefix"], item["key"]))
444 return result
446 # ------------------------------------------------------------------
447 # Email-specific methods
448 # ------------------------------------------------------------------
450 def _settings_to_gotenberg_pdfa(self) -> PdfAFormat | None:
451 """Convert the OCR output type setting to a Gotenberg PdfAFormat."""
452 if settings.OCR_OUTPUT_TYPE in {
453 OutputTypeChoices.PDF_A,
454 OutputTypeChoices.PDF_A2,
455 }:
456 return PdfAFormat.A2b
457 elif settings.OCR_OUTPUT_TYPE == OutputTypeChoices.PDF_A1: # pragma: no cover
458 logger.warning(
459 "Gotenberg does not support PDF/A-1a, choosing PDF/A-2b instead",
460 )
461 return PdfAFormat.A2b
462 elif settings.OCR_OUTPUT_TYPE == OutputTypeChoices.PDF_A3: # pragma: no cover
463 return PdfAFormat.A3b
464 return None
466 @staticmethod
467 def parse_file_to_message(filepath: Path) -> MailMessage:
468 """Parse the given .eml file into a MailMessage object.
470 Parameters
471 ----------
472 filepath:
473 Path to the .eml file.
475 Returns
476 -------
477 MailMessage
478 Parsed mail message.
480 Raises
481 ------
482 documents.parsers.ParseError
483 If the file cannot be parsed or is missing required fields.
484 """
485 try:
486 with filepath.open("rb") as eml:
487 parsed = MailMessage.from_bytes(eml.read())
488 if parsed.from_values is None:
489 raise ParseError(
490 f"Could not parse {filepath}: Missing 'from'",
491 )
492 except Exception as err:
493 raise ParseError(
494 f"Could not parse {filepath}: {err}",
495 ) from err
497 if is_naive(parsed.date):
498 parsed.date = make_aware(parsed.date)
500 return parsed
502 def tika_parse(self, html: str) -> str:
503 """Send HTML content to the Tika server for text extraction.
505 Parameters
506 ----------
507 html:
508 HTML string to parse.
510 Returns
511 -------
512 str
513 Extracted plain text.
515 Raises
516 ------
517 documents.parsers.ParseError
518 If the Tika server cannot be reached or returns an error.
519 """
520 logger.info("Sending content to Tika server")
522 try:
523 with TikaClient(tika_url=settings.TIKA_ENDPOINT) as client:
524 parsed = client.tika.as_text.from_buffer(html, "text/html")
526 if parsed.content is not None:
527 return parsed.content.strip()
528 return ""
529 except Exception as err:
530 raise ParseError(
531 f"Could not parse content with tika server at "
532 f"{settings.TIKA_ENDPOINT}: {err}",
533 ) from err
535 def generate_pdf(
536 self,
537 mail_message: MailMessage,
538 pdf_layout: MailRule.PdfLayout | None = None,
539 ) -> Path:
540 """Generate a PDF from the email message.
542 Creates separate PDFs for the email body and HTML content, then
543 merges them according to the requested layout.
545 Parameters
546 ----------
547 mail_message:
548 Parsed email message.
549 pdf_layout:
550 Layout option for the PDF. Falls back to the
551 EMAIL_PARSE_DEFAULT_LAYOUT setting if not provided.
553 Returns
554 -------
555 Path
556 Path to the generated PDF inside the temporary directory.
557 """
558 archive_path = Path(self._tempdir) / "merged.pdf"
560 mail_pdf_file = self.generate_pdf_from_mail(mail_message)
562 if pdf_layout is None:
563 pdf_layout = MailRule.PdfLayout(settings.EMAIL_PARSE_DEFAULT_LAYOUT)
565 # If no HTML content, create the PDF from the message.
566 # Otherwise, create 2 PDFs and merge them with Gotenberg.
567 if not mail_message.html:
568 archive_path.write_bytes(mail_pdf_file.read_bytes())
569 else:
570 pdf_of_html_content = self.generate_pdf_from_html(
571 mail_message.html,
572 mail_message.attachments,
573 )
575 logger.debug("Merging email text and HTML content into single PDF")
577 with (
578 GotenbergClient(
579 host=settings.TIKA_GOTENBERG_ENDPOINT,
580 timeout=settings.CELERY_TASK_TIME_LIMIT,
581 ) as client,
582 client.merge.merge() as route,
583 ):
584 # Configure requested PDF/A formatting, if any
585 pdf_a_format = self._settings_to_gotenberg_pdfa()
586 if pdf_a_format is not None:
587 route.pdf_format(pdf_a_format)
589 match pdf_layout:
590 case MailRule.PdfLayout.HTML_TEXT:
591 route.merge([pdf_of_html_content, mail_pdf_file])
592 case MailRule.PdfLayout.HTML_ONLY:
593 route.merge([pdf_of_html_content])
594 case MailRule.PdfLayout.TEXT_ONLY:
595 route.merge([mail_pdf_file])
596 case MailRule.PdfLayout.TEXT_HTML | _:
597 route.merge([mail_pdf_file, pdf_of_html_content])
599 try:
600 response = route.run()
601 archive_path.write_bytes(response.content)
602 except Exception as err:
603 raise ParseError(
604 f"Error while merging email HTML into PDF: {err}",
605 ) from err
607 return archive_path
609 def mail_to_html(self, mail: MailMessage) -> Path:
610 """Convert the given email into an HTML file using a template.
612 Parameters
613 ----------
614 mail:
615 Parsed mail message.
617 Returns
618 -------
619 Path
620 Path to the rendered HTML file inside the temporary directory.
621 """
623 def clean_html(text: str) -> str:
624 """Attempt to clean, escape, and linkify the given HTML string."""
625 if isinstance(text, list):
626 text = "\n".join([str(e) for e in text])
627 if not isinstance(text, str):
628 text = str(text)
629 text = escape(text)
630 text = clean(text)
631 text = linkify(text, Linkify(parse_email=True))
632 text = text.replace("\n", "<br>")
633 return text
635 data = {}
637 data["subject"] = clean_html(mail.subject)
638 if data["subject"]:
639 data["subject_label"] = "Subject"
640 data["from"] = clean_html(mail.from_values.full if mail.from_values else "")
641 if data["from"]:
642 data["from_label"] = "From"
643 data["to"] = clean_html(", ".join(address.full for address in mail.to_values))
644 if data["to"]:
645 data["to_label"] = "To"
646 data["cc"] = clean_html(", ".join(address.full for address in mail.cc_values))
647 if data["cc"]:
648 data["cc_label"] = "CC"
649 data["bcc"] = clean_html(", ".join(address.full for address in mail.bcc_values))
650 if data["bcc"]:
651 data["bcc_label"] = "BCC"
653 att = []
654 for a in mail.attachments:
655 att.append(
656 f"{a.filename} ({naturalsize(a.size, binary=True, format='%.2f')})",
657 )
658 data["attachments"] = clean_html(", ".join(att))
659 if data["attachments"]:
660 data["attachments_label"] = "Attachments"
662 data["date"] = clean_html(
663 timezone.localtime(mail.date).strftime("%Y-%m-%d %H:%M"),
664 )
665 data["content"] = clean_html(mail.text.strip())
667 from django.template.loader import render_to_string
669 html_file = Path(self._tempdir) / "email_as_html.html"
670 html_file.write_text(render_to_string("email_msg_template.html", context=data))
672 return html_file
674 def generate_pdf_from_mail(self, mail: MailMessage) -> Path:
675 """Create a PDF from the email body using an HTML template and Gotenberg.
677 Parameters
678 ----------
679 mail:
680 Parsed mail message.
682 Returns
683 -------
684 Path
685 Path to the generated PDF inside the temporary directory.
687 Raises
688 ------
689 documents.parsers.ParseError
690 If Gotenberg returns an error.
691 """
692 logger.info("Converting mail to PDF")
694 css_file = (
695 Path(__file__).parent.parent.parent
696 / "paperless_mail"
697 / "templates"
698 / "output.css"
699 )
700 email_html_file = self.mail_to_html(mail)
702 with (
703 GotenbergClient(
704 host=settings.TIKA_GOTENBERG_ENDPOINT,
705 timeout=settings.CELERY_TASK_TIME_LIMIT,
706 ) as client,
707 client.chromium.html_to_pdf() as route,
708 ):
709 # Configure requested PDF/A formatting, if any
710 pdf_a_format = self._settings_to_gotenberg_pdfa()
711 if pdf_a_format is not None:
712 route.pdf_format(pdf_a_format)
714 try:
715 response = (
716 route.index(email_html_file)
717 .resource(css_file)
718 .margins(
719 PageMarginsType(
720 top=Measurement(0.1, MeasurementUnitType.Inches),
721 bottom=Measurement(0.1, MeasurementUnitType.Inches),
722 left=Measurement(0.1, MeasurementUnitType.Inches),
723 right=Measurement(0.1, MeasurementUnitType.Inches),
724 ),
725 )
726 .size(A4)
727 .scale(1.0)
728 .run()
729 )
730 except Exception as err:
731 raise ParseError(
732 f"Error while converting email to PDF: {err}",
733 ) from err
735 email_as_pdf_file = Path(self._tempdir) / "email_as_pdf.pdf"
736 email_as_pdf_file.write_bytes(response.content)
738 return email_as_pdf_file
740 def generate_pdf_from_html(
741 self,
742 orig_html: str,
743 attachments: list[MailAttachment],
744 ) -> Path:
745 """Generate a PDF from the HTML content of the email.
747 Parameters
748 ----------
749 orig_html:
750 Raw HTML string from the email body.
751 attachments:
752 List of email attachments (used as inline resources).
754 Returns
755 -------
756 Path
757 Path to the generated PDF inside the temporary directory.
759 Raises
760 ------
761 documents.parsers.ParseError
762 If Gotenberg returns an error.
763 """
765 def clean_html_script(text: str) -> str:
766 compiled_open = re.compile(re.escape("<script"), re.IGNORECASE)
767 text = compiled_open.sub("<div hidden ", text)
769 compiled_close = re.compile(re.escape("</script"), re.IGNORECASE)
770 text = compiled_close.sub("</div", text)
771 return text
773 logger.info("Converting message html to PDF")
775 tempdir = Path(self._tempdir)
777 html_clean = clean_html_script(orig_html)
778 html_clean_file = tempdir / "index.html"
779 html_clean_file.write_text(html_clean)
781 with (
782 GotenbergClient(
783 host=settings.TIKA_GOTENBERG_ENDPOINT,
784 timeout=settings.CELERY_TASK_TIME_LIMIT,
785 ) as client,
786 client.chromium.html_to_pdf() as route,
787 ):
788 # Configure requested PDF/A formatting, if any
789 pdf_a_format = self._settings_to_gotenberg_pdfa()
790 if pdf_a_format is not None:
791 route.pdf_format(pdf_a_format)
793 # Add attachments as resources, cleaning the filename and replacing
794 # it in the index file for inclusion
795 for attachment in attachments:
796 # Clean the attachment name to be valid
797 name_cid = f"cid:{attachment.content_id}"
798 name_clean = "".join(e for e in name_cid if e.isalnum())
800 # Write attachment payload to a temp file
801 temp_file = tempdir / name_clean
802 temp_file.write_bytes(attachment.payload)
804 route.resource(temp_file)
806 # Replace as needed the name with the clean name
807 html_clean = html_clean.replace(name_cid, name_clean)
809 # Now store the cleaned up HTML version
810 html_clean_file = tempdir / "index.html"
811 html_clean_file.write_text(html_clean)
812 # This is our index file, the main page basically
813 route.index(html_clean_file)
815 # Set page size, margins
816 route.margins(
817 PageMarginsType(
818 top=Measurement(0.1, MeasurementUnitType.Inches),
819 bottom=Measurement(0.1, MeasurementUnitType.Inches),
820 left=Measurement(0.1, MeasurementUnitType.Inches),
821 right=Measurement(0.1, MeasurementUnitType.Inches),
822 ),
823 ).size(A4).scale(1.0)
825 try:
826 response = route.run()
828 except Exception as err:
829 raise ParseError(
830 f"Error while converting document to PDF: {err}",
831 ) from err
833 html_pdf = tempdir / "html.pdf"
834 html_pdf.write_bytes(response.content)
835 return html_pdf