Coverage for paperless/parsers/mail.py: 19%

286 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1""" 

2Built-in mail document parser. 

3 

4Handles message/rfc822 (EML) MIME type by: 

5- Parsing the email using imap_tools 

6- Generating a PDF via Gotenberg (for display and archive) 

7- Extracting text via Tika for HTML content 

8- Extracting metadata from email headers 

9 

10The parser always produces a PDF because EML files cannot be rendered 

11natively in a browser (requires_pdf_rendition=True). 

12""" 

13 

14from __future__ import annotations 

15 

16import logging 

17import re 

18import shutil 

19import tempfile 

20from html import escape 

21from pathlib import Path 

22from typing import TYPE_CHECKING 

23from typing import Self 

24 

25from django.conf import settings 

26from django.utils import timezone 

27from django.utils.timezone import is_naive 

28from django.utils.timezone import make_aware 

29from gotenberg_client import GotenbergClient 

30from gotenberg_client.constants import A4 

31from gotenberg_client.options import Measurement 

32from gotenberg_client.options import MeasurementUnitType 

33from gotenberg_client.options import PageMarginsType 

34from gotenberg_client.options import PdfAFormat 

35from humanize import naturalsize 

36from imap_tools import MailAttachment 

37from imap_tools import MailMessage 

38from tika_client import TikaClient 

39from turbohtml.clean import Linkify 

40from turbohtml.clean import linkify 

41from turbohtml.migration.bleach import clean 

42 

43from documents.parsers import ParseError 

44from documents.parsers import make_thumbnail_from_pdf 

45from paperless.models import OutputTypeChoices 

46from paperless.version import __full_version_str__ 

47from paperless_mail.models import MailRule 

48 

49if TYPE_CHECKING: 49 ↛ 50line 49 didn't jump to line 50 because the condition on line 49 was never true

50 import datetime 

51 from types import TracebackType 

52 

53 from paperless.parsers import MetadataEntry 

54 from paperless.parsers import ParserContext 

55 

56logger = logging.getLogger("paperless.parsing.mail") 

57 

58_SUPPORTED_MIME_TYPES: dict[str, str] = { 

59 "message/rfc822": ".eml", 

60} 

61 

62 

63class MailDocumentParser: 

64 """Parse .eml email files for Paperless-ngx. 

65 

66 Uses imap_tools to parse .eml files, generates a PDF using Gotenberg, 

67 and sends the HTML part to a Tika server for text extraction. Because 

68 EML files cannot be rendered natively in a browser, the parser always 

69 produces a PDF rendition (requires_pdf_rendition=True). 

70 

71 Pass a ``ParserContext`` to ``configure()`` before ``parse()`` to 

72 apply mail-rule-specific PDF layout options: 

73 

74 parser.configure(ParserContext(mailrule_id=rule.pk)) 

75 parser.parse(path, mime_type) 

76 

77 Class attributes 

78 ---------------- 

79 name : str 

80 Human-readable parser name. 

81 version : str 

82 Semantic version string, kept in sync with Paperless-ngx releases. 

83 author : str 

84 Maintainer name. 

85 url : str 

86 Issue tracker / source URL. 

87 """ 

88 

89 name: str = "Paperless-ngx Mail Parser" 

90 version: str = __full_version_str__ 

91 author: str = "Paperless-ngx Contributors" 

92 url: str = "https://github.com/paperless-ngx/paperless-ngx" 

93 

94 # ------------------------------------------------------------------ 

95 # Class methods 

96 # ------------------------------------------------------------------ 

97 

98 @classmethod 

99 def supported_mime_types(cls) -> dict[str, str]: 

100 """Return the MIME types this parser handles. 

101 

102 Returns 

103 ------- 

104 dict[str, str] 

105 Mapping of MIME type to preferred file extension. 

106 """ 

107 return _SUPPORTED_MIME_TYPES 

108 

109 @classmethod 

110 def score( 

111 cls, 

112 mime_type: str, 

113 filename: str, 

114 path: Path | None = None, 

115 ) -> int | None: 

116 """Return the priority score for handling this file. 

117 

118 Parameters 

119 ---------- 

120 mime_type: 

121 Detected MIME type of the file. 

122 filename: 

123 Original filename including extension. 

124 path: 

125 Optional filesystem path. Not inspected by this parser. 

126 

127 Returns 

128 ------- 

129 int | None 

130 10 if the MIME type is supported, otherwise None. 

131 """ 

132 if mime_type in _SUPPORTED_MIME_TYPES: 

133 return 10 

134 return None 

135 

136 # ------------------------------------------------------------------ 

137 # Properties 

138 # ------------------------------------------------------------------ 

139 

140 @property 

141 def can_produce_archive(self) -> bool: 

142 """Whether this parser can produce a searchable PDF archive copy. 

143 

144 Returns 

145 ------- 

146 bool 

147 Always False — the mail parser produces a display PDF 

148 (requires_pdf_rendition=True), not an optional OCR archive. 

149 """ 

150 return False 

151 

152 @property 

153 def requires_pdf_rendition(self) -> bool: 

154 """Whether the parser must produce a PDF for the frontend to display. 

155 

156 Returns 

157 ------- 

158 bool 

159 Always True — EML files cannot be rendered natively in a browser, 

160 so a PDF conversion is always required for display. 

161 """ 

162 return True 

163 

164 # ------------------------------------------------------------------ 

165 # Lifecycle 

166 # ------------------------------------------------------------------ 

167 

168 def __init__(self, logging_group: object = None) -> None: 

169 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True) 

170 self._tempdir = Path( 

171 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR), 

172 ) 

173 self._text: str | None = None 

174 self._date: datetime.datetime | None = None 

175 self._archive_path: Path | None = None 

176 self._mailrule_id: int | None = None 

177 

178 def __enter__(self) -> Self: 

179 return self 

180 

181 def __exit__( 

182 self, 

183 exc_type: type[BaseException] | None, 

184 exc_val: BaseException | None, 

185 exc_tb: TracebackType | None, 

186 ) -> None: 

187 logger.debug("Cleaning up temporary directory %s", self._tempdir) 

188 shutil.rmtree(self._tempdir, ignore_errors=True) 

189 

190 # ------------------------------------------------------------------ 

191 # Core parsing interface 

192 # ------------------------------------------------------------------ 

193 

194 def configure(self, context: ParserContext) -> None: 

195 self._mailrule_id = context.mailrule_id 

196 

197 def parse( 

198 self, 

199 document_path: Path, 

200 mime_type: str, 

201 *, 

202 produce_archive: bool = True, 

203 ) -> None: 

204 """Parse the given .eml into formatted text and a PDF archive. 

205 

206 Call ``configure(ParserContext(mailrule_id=...))`` before this method 

207 to apply mail-rule-specific PDF layout options. The ``produce_archive`` 

208 flag is accepted for protocol compatibility but is always honoured — 

209 the mail parser always produces a PDF since EML files cannot be 

210 displayed natively. 

211 

212 Parameters 

213 ---------- 

214 document_path: 

215 Absolute path to the .eml file. 

216 mime_type: 

217 Detected MIME type of the document (should be "message/rfc822"). 

218 produce_archive: 

219 Accepted for protocol compatibility. The PDF rendition is always 

220 produced since EML files cannot be displayed natively in a browser. 

221 

222 Raises 

223 ------ 

224 documents.parsers.ParseError 

225 If the file cannot be parsed or PDF generation fails. 

226 """ 

227 

228 def strip_text(text: str) -> str: 

229 """Reduces the spacing of the given text string.""" 

230 text = re.sub(r"\s+", " ", text) 

231 text = re.sub(r"(\n *)+", "\n", text) 

232 return text.strip() 

233 

234 def build_formatted_text(mail_message: MailMessage) -> str: 

235 """Constructs a formatted string based on the given email.""" 

236 fmt_text = f"Subject: {mail_message.subject}\n\n" 

237 fmt_text += f"From: {mail_message.from_values.full if mail_message.from_values else ''}\n\n" 

238 to_list = [address.full for address in mail_message.to_values] 

239 fmt_text += f"To: {', '.join(to_list)}\n\n" 

240 if mail_message.cc_values: 

241 fmt_text += ( 

242 f"CC: {', '.join(address.full for address in mail.cc_values)}\n\n" 

243 ) 

244 if mail_message.bcc_values: 

245 fmt_text += ( 

246 f"BCC: {', '.join(address.full for address in mail.bcc_values)}\n\n" 

247 ) 

248 if mail_message.attachments: 

249 att = [] 

250 for a in mail.attachments: 

251 attachment_size = naturalsize(a.size, binary=True, format="%.2f") 

252 att.append( 

253 f"{a.filename} ({attachment_size})", 

254 ) 

255 fmt_text += f"Attachments: {', '.join(att)}\n\n" 

256 

257 if mail.html: 

258 fmt_text += "HTML content: " + strip_text(self.tika_parse(mail.html)) 

259 

260 fmt_text += f"\n\n{strip_text(mail.text)}" 

261 

262 return fmt_text 

263 

264 logger.debug("Parsing file %s into an email", document_path.name) 

265 mail = self.parse_file_to_message(document_path) 

266 

267 logger.debug("Building formatted text from email") 

268 self._text = build_formatted_text(mail) 

269 

270 self._date = mail.date 

271 

272 logger.debug("Creating a PDF from the email") 

273 if self._mailrule_id: 

274 rule = MailRule.objects.get(pk=self._mailrule_id) 

275 self._archive_path = self.generate_pdf( 

276 mail, 

277 MailRule.PdfLayout(rule.pdf_layout), 

278 ) 

279 else: 

280 self._archive_path = self.generate_pdf(mail) 

281 

282 # ------------------------------------------------------------------ 

283 # Result accessors 

284 # ------------------------------------------------------------------ 

285 

286 def get_text(self) -> str: 

287 """Return the plain-text content extracted during parse. 

288 

289 Returns 

290 ------- 

291 str 

292 Extracted text, or an empty string if no text could be found. 

293 """ 

294 return self._text or "" 

295 

296 def get_date(self) -> datetime.datetime | None: 

297 """Return the document date detected during parse. 

298 

299 Returns 

300 ------- 

301 datetime.datetime | None 

302 Date from the email headers, or None if not detected. 

303 """ 

304 return self._date 

305 

306 def get_archive_path(self) -> Path | None: 

307 """Return the path to the generated archive PDF, or None. 

308 

309 Returns 

310 ------- 

311 Path | None 

312 Path to the PDF produced by Gotenberg, or None if parse has not 

313 been called yet. 

314 """ 

315 return self._archive_path 

316 

317 # ------------------------------------------------------------------ 

318 # Thumbnail and metadata 

319 # ------------------------------------------------------------------ 

320 

321 def get_thumbnail( 

322 self, 

323 document_path: Path, 

324 mime_type: str, 

325 file_name: str | None = None, 

326 ) -> Path: 

327 """Generate a thumbnail from the PDF rendition of the email. 

328 

329 Converts the document to PDF first if not already done. 

330 

331 Parameters 

332 ---------- 

333 document_path: 

334 Absolute path to the source document. 

335 mime_type: 

336 Detected MIME type of the document. 

337 file_name: 

338 Kept for backward compatibility; not used. 

339 

340 Returns 

341 ------- 

342 Path 

343 Path to the generated WebP thumbnail inside the temporary directory. 

344 """ 

345 if not self._archive_path: 

346 self._archive_path = self.generate_pdf( 

347 self.parse_file_to_message(document_path), 

348 ) 

349 

350 return make_thumbnail_from_pdf( 

351 self._archive_path, 

352 self._tempdir, 

353 ) 

354 

355 def get_page_count( 

356 self, 

357 document_path: Path, 

358 mime_type: str, 

359 ) -> int | None: 

360 """Return the number of pages in the document. 

361 

362 Counts pages in the archive PDF produced by a preceding parse() 

363 call. Returns ``None`` if parse() has not been called yet or if 

364 no archive was produced. 

365 

366 Returns 

367 ------- 

368 int | None 

369 Page count of the archive PDF, or ``None``. 

370 """ 

371 if self._archive_path is not None: 

372 from paperless.parsers.utils import get_page_count_for_pdf 

373 

374 return get_page_count_for_pdf(self._archive_path, log=logger) 

375 return None 

376 

377 def extract_metadata( 

378 self, 

379 document_path: Path, 

380 mime_type: str, 

381 ) -> list[MetadataEntry]: 

382 """Extract metadata from the email headers. 

383 

384 Returns email headers as metadata entries with prefix "header", 

385 plus summary entries for attachments and date. 

386 

387 Returns 

388 ------- 

389 list[MetadataEntry] 

390 Sorted list of metadata entries, or ``[]`` on parse failure. 

391 """ 

392 result: list[MetadataEntry] = [] 

393 

394 try: 

395 mail = self.parse_file_to_message(document_path) 

396 except ParseError as e: 

397 logger.warning( 

398 "Error while fetching document metadata for %s: %s", 

399 document_path, 

400 e, 

401 ) 

402 return result 

403 

404 for key, header_values in mail.headers.items(): 

405 value = ", ".join(header_values) 

406 try: 

407 value.encode("utf-8") 

408 except UnicodeEncodeError as e: # pragma: no cover 

409 logger.debug("Skipping header %s: %s", key, e) 

410 continue 

411 

412 result.append( 

413 { 

414 "namespace": "", 

415 "prefix": "header", 

416 "key": key, 

417 "value": value, 

418 }, 

419 ) 

420 

421 result.append( 

422 { 

423 "namespace": "", 

424 "prefix": "", 

425 "key": "attachments", 

426 "value": ", ".join( 

427 f"{attachment.filename}" 

428 f"({naturalsize(attachment.size, binary=True, format='%.2f')})" 

429 for attachment in mail.attachments 

430 ), 

431 }, 

432 ) 

433 

434 result.append( 

435 { 

436 "namespace": "", 

437 "prefix": "", 

438 "key": "date", 

439 "value": mail.date.strftime("%Y-%m-%d %H:%M:%S %Z"), 

440 }, 

441 ) 

442 

443 result.sort(key=lambda item: (item["prefix"], item["key"])) 

444 return result 

445 

446 # ------------------------------------------------------------------ 

447 # Email-specific methods 

448 # ------------------------------------------------------------------ 

449 

450 def _settings_to_gotenberg_pdfa(self) -> PdfAFormat | None: 

451 """Convert the OCR output type setting to a Gotenberg PdfAFormat.""" 

452 if settings.OCR_OUTPUT_TYPE in { 

453 OutputTypeChoices.PDF_A, 

454 OutputTypeChoices.PDF_A2, 

455 }: 

456 return PdfAFormat.A2b 

457 elif settings.OCR_OUTPUT_TYPE == OutputTypeChoices.PDF_A1: # pragma: no cover 

458 logger.warning( 

459 "Gotenberg does not support PDF/A-1a, choosing PDF/A-2b instead", 

460 ) 

461 return PdfAFormat.A2b 

462 elif settings.OCR_OUTPUT_TYPE == OutputTypeChoices.PDF_A3: # pragma: no cover 

463 return PdfAFormat.A3b 

464 return None 

465 

466 @staticmethod 

467 def parse_file_to_message(filepath: Path) -> MailMessage: 

468 """Parse the given .eml file into a MailMessage object. 

469 

470 Parameters 

471 ---------- 

472 filepath: 

473 Path to the .eml file. 

474 

475 Returns 

476 ------- 

477 MailMessage 

478 Parsed mail message. 

479 

480 Raises 

481 ------ 

482 documents.parsers.ParseError 

483 If the file cannot be parsed or is missing required fields. 

484 """ 

485 try: 

486 with filepath.open("rb") as eml: 

487 parsed = MailMessage.from_bytes(eml.read()) 

488 if parsed.from_values is None: 

489 raise ParseError( 

490 f"Could not parse {filepath}: Missing 'from'", 

491 ) 

492 except Exception as err: 

493 raise ParseError( 

494 f"Could not parse {filepath}: {err}", 

495 ) from err 

496 

497 if is_naive(parsed.date): 

498 parsed.date = make_aware(parsed.date) 

499 

500 return parsed 

501 

502 def tika_parse(self, html: str) -> str: 

503 """Send HTML content to the Tika server for text extraction. 

504 

505 Parameters 

506 ---------- 

507 html: 

508 HTML string to parse. 

509 

510 Returns 

511 ------- 

512 str 

513 Extracted plain text. 

514 

515 Raises 

516 ------ 

517 documents.parsers.ParseError 

518 If the Tika server cannot be reached or returns an error. 

519 """ 

520 logger.info("Sending content to Tika server") 

521 

522 try: 

523 with TikaClient(tika_url=settings.TIKA_ENDPOINT) as client: 

524 parsed = client.tika.as_text.from_buffer(html, "text/html") 

525 

526 if parsed.content is not None: 

527 return parsed.content.strip() 

528 return "" 

529 except Exception as err: 

530 raise ParseError( 

531 f"Could not parse content with tika server at " 

532 f"{settings.TIKA_ENDPOINT}: {err}", 

533 ) from err 

534 

535 def generate_pdf( 

536 self, 

537 mail_message: MailMessage, 

538 pdf_layout: MailRule.PdfLayout | None = None, 

539 ) -> Path: 

540 """Generate a PDF from the email message. 

541 

542 Creates separate PDFs for the email body and HTML content, then 

543 merges them according to the requested layout. 

544 

545 Parameters 

546 ---------- 

547 mail_message: 

548 Parsed email message. 

549 pdf_layout: 

550 Layout option for the PDF. Falls back to the 

551 EMAIL_PARSE_DEFAULT_LAYOUT setting if not provided. 

552 

553 Returns 

554 ------- 

555 Path 

556 Path to the generated PDF inside the temporary directory. 

557 """ 

558 archive_path = Path(self._tempdir) / "merged.pdf" 

559 

560 mail_pdf_file = self.generate_pdf_from_mail(mail_message) 

561 

562 if pdf_layout is None: 

563 pdf_layout = MailRule.PdfLayout(settings.EMAIL_PARSE_DEFAULT_LAYOUT) 

564 

565 # If no HTML content, create the PDF from the message. 

566 # Otherwise, create 2 PDFs and merge them with Gotenberg. 

567 if not mail_message.html: 

568 archive_path.write_bytes(mail_pdf_file.read_bytes()) 

569 else: 

570 pdf_of_html_content = self.generate_pdf_from_html( 

571 mail_message.html, 

572 mail_message.attachments, 

573 ) 

574 

575 logger.debug("Merging email text and HTML content into single PDF") 

576 

577 with ( 

578 GotenbergClient( 

579 host=settings.TIKA_GOTENBERG_ENDPOINT, 

580 timeout=settings.CELERY_TASK_TIME_LIMIT, 

581 ) as client, 

582 client.merge.merge() as route, 

583 ): 

584 # Configure requested PDF/A formatting, if any 

585 pdf_a_format = self._settings_to_gotenberg_pdfa() 

586 if pdf_a_format is not None: 

587 route.pdf_format(pdf_a_format) 

588 

589 match pdf_layout: 

590 case MailRule.PdfLayout.HTML_TEXT: 

591 route.merge([pdf_of_html_content, mail_pdf_file]) 

592 case MailRule.PdfLayout.HTML_ONLY: 

593 route.merge([pdf_of_html_content]) 

594 case MailRule.PdfLayout.TEXT_ONLY: 

595 route.merge([mail_pdf_file]) 

596 case MailRule.PdfLayout.TEXT_HTML | _: 

597 route.merge([mail_pdf_file, pdf_of_html_content]) 

598 

599 try: 

600 response = route.run() 

601 archive_path.write_bytes(response.content) 

602 except Exception as err: 

603 raise ParseError( 

604 f"Error while merging email HTML into PDF: {err}", 

605 ) from err 

606 

607 return archive_path 

608 

609 def mail_to_html(self, mail: MailMessage) -> Path: 

610 """Convert the given email into an HTML file using a template. 

611 

612 Parameters 

613 ---------- 

614 mail: 

615 Parsed mail message. 

616 

617 Returns 

618 ------- 

619 Path 

620 Path to the rendered HTML file inside the temporary directory. 

621 """ 

622 

623 def clean_html(text: str) -> str: 

624 """Attempt to clean, escape, and linkify the given HTML string.""" 

625 if isinstance(text, list): 

626 text = "\n".join([str(e) for e in text]) 

627 if not isinstance(text, str): 

628 text = str(text) 

629 text = escape(text) 

630 text = clean(text) 

631 text = linkify(text, Linkify(parse_email=True)) 

632 text = text.replace("\n", "<br>") 

633 return text 

634 

635 data = {} 

636 

637 data["subject"] = clean_html(mail.subject) 

638 if data["subject"]: 

639 data["subject_label"] = "Subject" 

640 data["from"] = clean_html(mail.from_values.full if mail.from_values else "") 

641 if data["from"]: 

642 data["from_label"] = "From" 

643 data["to"] = clean_html(", ".join(address.full for address in mail.to_values)) 

644 if data["to"]: 

645 data["to_label"] = "To" 

646 data["cc"] = clean_html(", ".join(address.full for address in mail.cc_values)) 

647 if data["cc"]: 

648 data["cc_label"] = "CC" 

649 data["bcc"] = clean_html(", ".join(address.full for address in mail.bcc_values)) 

650 if data["bcc"]: 

651 data["bcc_label"] = "BCC" 

652 

653 att = [] 

654 for a in mail.attachments: 

655 att.append( 

656 f"{a.filename} ({naturalsize(a.size, binary=True, format='%.2f')})", 

657 ) 

658 data["attachments"] = clean_html(", ".join(att)) 

659 if data["attachments"]: 

660 data["attachments_label"] = "Attachments" 

661 

662 data["date"] = clean_html( 

663 timezone.localtime(mail.date).strftime("%Y-%m-%d %H:%M"), 

664 ) 

665 data["content"] = clean_html(mail.text.strip()) 

666 

667 from django.template.loader import render_to_string 

668 

669 html_file = Path(self._tempdir) / "email_as_html.html" 

670 html_file.write_text(render_to_string("email_msg_template.html", context=data)) 

671 

672 return html_file 

673 

674 def generate_pdf_from_mail(self, mail: MailMessage) -> Path: 

675 """Create a PDF from the email body using an HTML template and Gotenberg. 

676 

677 Parameters 

678 ---------- 

679 mail: 

680 Parsed mail message. 

681 

682 Returns 

683 ------- 

684 Path 

685 Path to the generated PDF inside the temporary directory. 

686 

687 Raises 

688 ------ 

689 documents.parsers.ParseError 

690 If Gotenberg returns an error. 

691 """ 

692 logger.info("Converting mail to PDF") 

693 

694 css_file = ( 

695 Path(__file__).parent.parent.parent 

696 / "paperless_mail" 

697 / "templates" 

698 / "output.css" 

699 ) 

700 email_html_file = self.mail_to_html(mail) 

701 

702 with ( 

703 GotenbergClient( 

704 host=settings.TIKA_GOTENBERG_ENDPOINT, 

705 timeout=settings.CELERY_TASK_TIME_LIMIT, 

706 ) as client, 

707 client.chromium.html_to_pdf() as route, 

708 ): 

709 # Configure requested PDF/A formatting, if any 

710 pdf_a_format = self._settings_to_gotenberg_pdfa() 

711 if pdf_a_format is not None: 

712 route.pdf_format(pdf_a_format) 

713 

714 try: 

715 response = ( 

716 route.index(email_html_file) 

717 .resource(css_file) 

718 .margins( 

719 PageMarginsType( 

720 top=Measurement(0.1, MeasurementUnitType.Inches), 

721 bottom=Measurement(0.1, MeasurementUnitType.Inches), 

722 left=Measurement(0.1, MeasurementUnitType.Inches), 

723 right=Measurement(0.1, MeasurementUnitType.Inches), 

724 ), 

725 ) 

726 .size(A4) 

727 .scale(1.0) 

728 .run() 

729 ) 

730 except Exception as err: 

731 raise ParseError( 

732 f"Error while converting email to PDF: {err}", 

733 ) from err 

734 

735 email_as_pdf_file = Path(self._tempdir) / "email_as_pdf.pdf" 

736 email_as_pdf_file.write_bytes(response.content) 

737 

738 return email_as_pdf_file 

739 

740 def generate_pdf_from_html( 

741 self, 

742 orig_html: str, 

743 attachments: list[MailAttachment], 

744 ) -> Path: 

745 """Generate a PDF from the HTML content of the email. 

746 

747 Parameters 

748 ---------- 

749 orig_html: 

750 Raw HTML string from the email body. 

751 attachments: 

752 List of email attachments (used as inline resources). 

753 

754 Returns 

755 ------- 

756 Path 

757 Path to the generated PDF inside the temporary directory. 

758 

759 Raises 

760 ------ 

761 documents.parsers.ParseError 

762 If Gotenberg returns an error. 

763 """ 

764 

765 def clean_html_script(text: str) -> str: 

766 compiled_open = re.compile(re.escape("<script"), re.IGNORECASE) 

767 text = compiled_open.sub("<div hidden ", text) 

768 

769 compiled_close = re.compile(re.escape("</script"), re.IGNORECASE) 

770 text = compiled_close.sub("</div", text) 

771 return text 

772 

773 logger.info("Converting message html to PDF") 

774 

775 tempdir = Path(self._tempdir) 

776 

777 html_clean = clean_html_script(orig_html) 

778 html_clean_file = tempdir / "index.html" 

779 html_clean_file.write_text(html_clean) 

780 

781 with ( 

782 GotenbergClient( 

783 host=settings.TIKA_GOTENBERG_ENDPOINT, 

784 timeout=settings.CELERY_TASK_TIME_LIMIT, 

785 ) as client, 

786 client.chromium.html_to_pdf() as route, 

787 ): 

788 # Configure requested PDF/A formatting, if any 

789 pdf_a_format = self._settings_to_gotenberg_pdfa() 

790 if pdf_a_format is not None: 

791 route.pdf_format(pdf_a_format) 

792 

793 # Add attachments as resources, cleaning the filename and replacing 

794 # it in the index file for inclusion 

795 for attachment in attachments: 

796 # Clean the attachment name to be valid 

797 name_cid = f"cid:{attachment.content_id}" 

798 name_clean = "".join(e for e in name_cid if e.isalnum()) 

799 

800 # Write attachment payload to a temp file 

801 temp_file = tempdir / name_clean 

802 temp_file.write_bytes(attachment.payload) 

803 

804 route.resource(temp_file) 

805 

806 # Replace as needed the name with the clean name 

807 html_clean = html_clean.replace(name_cid, name_clean) 

808 

809 # Now store the cleaned up HTML version 

810 html_clean_file = tempdir / "index.html" 

811 html_clean_file.write_text(html_clean) 

812 # This is our index file, the main page basically 

813 route.index(html_clean_file) 

814 

815 # Set page size, margins 

816 route.margins( 

817 PageMarginsType( 

818 top=Measurement(0.1, MeasurementUnitType.Inches), 

819 bottom=Measurement(0.1, MeasurementUnitType.Inches), 

820 left=Measurement(0.1, MeasurementUnitType.Inches), 

821 right=Measurement(0.1, MeasurementUnitType.Inches), 

822 ), 

823 ).size(A4).scale(1.0) 

824 

825 try: 

826 response = route.run() 

827 

828 except Exception as err: 

829 raise ParseError( 

830 f"Error while converting document to PDF: {err}", 

831 ) from err 

832 

833 html_pdf = tempdir / "html.pdf" 

834 html_pdf.write_bytes(response.content) 

835 return html_pdf