Coverage for paperless/parsers/tika.py: 31%
130 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1"""
2Built-in Tika document parser.
4Handles Office documents (DOCX, ODT, XLS, XLSX, PPT, PPTX, RTF, etc.) by
5sending them to an Apache Tika server for text extraction and a Gotenberg
6server for PDF conversion. Because the source formats cannot be rendered by
7a browser natively, the parser always produces a PDF rendition for display.
8"""
10from __future__ import annotations
12import logging
13import shutil
14import tempfile
15from contextlib import ExitStack
16from pathlib import Path
17from typing import TYPE_CHECKING
18from typing import Self
20import httpx
21from django.conf import settings
22from django.utils import timezone
23from gotenberg_client import GotenbergClient
24from gotenberg_client.options import PdfAFormat
25from tika_client import TikaClient
27from documents.parsers import ParseError
28from documents.parsers import make_thumbnail_from_pdf
29from paperless.config import OutputTypeConfig
30from paperless.models import OutputTypeChoices
31from paperless.version import __full_version_str__
33if TYPE_CHECKING: 33 ↛ 34line 33 didn't jump to line 34 because the condition on line 33 was never true
34 import datetime
35 from types import TracebackType
37 from paperless.parsers import MetadataEntry
38 from paperless.parsers import ParserContext
40logger = logging.getLogger("paperless.parsing.tika")
42_SUPPORTED_MIME_TYPES: dict[str, str] = {
43 "application/msword": ".doc",
44 "application/vnd.openxmlformats-officedocument.wordprocessingml.document": ".docx",
45 "application/vnd.ms-excel": ".xls",
46 "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": ".xlsx",
47 "application/vnd.ms-powerpoint": ".ppt",
48 "application/vnd.openxmlformats-officedocument.presentationml.presentation": ".pptx",
49 "application/vnd.openxmlformats-officedocument.presentationml.slideshow": ".ppsx",
50 "application/vnd.oasis.opendocument.presentation": ".odp",
51 "application/vnd.oasis.opendocument.spreadsheet": ".ods",
52 "application/vnd.oasis.opendocument.text": ".odt",
53 "application/vnd.oasis.opendocument.graphics": ".odg",
54 "text/rtf": ".rtf",
55}
58class TikaDocumentParser:
59 """Parse Office documents via Apache Tika and Gotenberg for Paperless-ngx.
61 Text extraction is handled by the Tika server. PDF conversion for display
62 is handled by Gotenberg (LibreOffice route). Because the source formats
63 cannot be rendered by a browser natively, ``requires_pdf_rendition`` is
64 True and the PDF is always produced regardless of the ``produce_archive``
65 flag passed to ``parse``.
67 Both ``TikaClient`` and ``GotenbergClient`` are opened once in
68 ``__enter__`` via an ``ExitStack`` and shared across ``parse``,
69 ``extract_metadata``, and ``_convert_to_pdf`` calls, then closed via
70 ``ExitStack.close()`` in ``__exit__``. The parser must always be used
71 as a context manager.
73 Class attributes
74 ----------------
75 name : str
76 Human-readable parser name.
77 version : str
78 Semantic version string, kept in sync with Paperless-ngx releases.
79 author : str
80 Maintainer name.
81 url : str
82 Issue tracker / source URL.
83 """
85 name: str = "Paperless-ngx Tika Parser"
86 version: str = __full_version_str__
87 author: str = "Paperless-ngx Contributors"
88 url: str = "https://github.com/paperless-ngx/paperless-ngx"
90 # ------------------------------------------------------------------
91 # Class methods
92 # ------------------------------------------------------------------
94 @classmethod
95 def supported_mime_types(cls) -> dict[str, str]:
96 """Return the MIME types this parser handles.
98 Returns
99 -------
100 dict[str, str]
101 Mapping of MIME type to preferred file extension.
102 """
103 return _SUPPORTED_MIME_TYPES
105 @classmethod
106 def score(
107 cls,
108 mime_type: str,
109 filename: str,
110 path: Path | None = None,
111 ) -> int | None:
112 """Return the priority score for handling this file.
114 Returns ``None`` when Tika integration is disabled so the registry
115 skips this parser entirely.
117 Parameters
118 ----------
119 mime_type:
120 Detected MIME type of the file.
121 filename:
122 Original filename including extension.
123 path:
124 Optional filesystem path. Not inspected by this parser.
126 Returns
127 -------
128 int | None
129 10 if TIKA_ENABLED and the MIME type is supported, otherwise None.
130 """
131 if not settings.TIKA_ENABLED:
132 return None
133 if mime_type in _SUPPORTED_MIME_TYPES:
134 return 10
135 return None
137 # ------------------------------------------------------------------
138 # Properties
139 # ------------------------------------------------------------------
141 @property
142 def can_produce_archive(self) -> bool:
143 """Whether this parser can produce a searchable PDF archive copy.
145 Returns
146 -------
147 bool
148 Always False — Tika produces a display PDF, not an OCR archive.
149 """
150 return False
152 @property
153 def requires_pdf_rendition(self) -> bool:
154 """Whether the parser must produce a PDF for the frontend to display.
156 Returns
157 -------
158 bool
159 Always True — Office formats cannot be rendered natively in a
160 browser, so a PDF conversion is always required for display.
161 """
162 return True
164 # ------------------------------------------------------------------
165 # Lifecycle
166 # ------------------------------------------------------------------
168 def __init__(self, logging_group: object = None) -> None:
169 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True)
170 self._tempdir = Path(
171 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR),
172 )
173 self._text: str | None = None
174 self._date: datetime.datetime | None = None
175 self._archive_path: Path | None = None
176 self._exit_stack = ExitStack()
177 self._tika_client: TikaClient | None = None
178 self._gotenberg_client: GotenbergClient | None = None
180 def __enter__(self) -> Self:
181 self._tika_client = self._exit_stack.enter_context(
182 TikaClient(
183 tika_url=settings.TIKA_ENDPOINT,
184 timeout=settings.CELERY_TASK_TIME_LIMIT,
185 ),
186 )
187 self._gotenberg_client = self._exit_stack.enter_context(
188 GotenbergClient(
189 host=settings.TIKA_GOTENBERG_ENDPOINT,
190 timeout=settings.CELERY_TASK_TIME_LIMIT,
191 ),
192 )
193 return self
195 def __exit__(
196 self,
197 exc_type: type[BaseException] | None,
198 exc_val: BaseException | None,
199 exc_tb: TracebackType | None,
200 ) -> None:
201 self._exit_stack.close()
202 logger.debug("Cleaning up temporary directory %s", self._tempdir)
203 shutil.rmtree(self._tempdir, ignore_errors=True)
205 # ------------------------------------------------------------------
206 # Core parsing interface
207 # ------------------------------------------------------------------
209 def configure(self, context: ParserContext) -> None:
210 pass
212 def parse(
213 self,
214 document_path: Path,
215 mime_type: str,
216 *,
217 produce_archive: bool = True,
218 ) -> None:
219 """Send the document to Tika for text extraction and Gotenberg for PDF.
221 Because ``requires_pdf_rendition`` is True the PDF conversion is
222 always performed — the ``produce_archive`` flag is intentionally
223 ignored.
225 Parameters
226 ----------
227 document_path:
228 Absolute path to the document file to parse.
229 mime_type:
230 Detected MIME type of the document.
231 produce_archive:
232 Accepted for protocol compatibility but ignored; the PDF rendition
233 is always produced since the source format cannot be displayed
234 natively in the browser.
236 Raises
237 ------
238 documents.parsers.ParseError
239 If Tika or Gotenberg returns an error.
240 """
241 if TYPE_CHECKING:
242 assert self._tika_client is not None
244 logger.info("Sending %s to Tika server", document_path)
246 try:
247 try:
248 parsed = self._tika_client.tika.as_text.from_file(
249 document_path,
250 mime_type,
251 )
252 except httpx.HTTPStatusError as err:
253 # Workaround https://issues.apache.org/jira/browse/TIKA-4110
254 # Tika fails with some files as multi-part form data
255 if err.response.status_code == httpx.codes.INTERNAL_SERVER_ERROR:
256 parsed = self._tika_client.tika.as_text.from_buffer(
257 document_path.read_bytes(),
258 mime_type,
259 )
260 else: # pragma: no cover
261 raise
262 except Exception as err:
263 raise ParseError(
264 f"Could not parse {document_path} with tika server at "
265 f"{settings.TIKA_ENDPOINT}: {err}",
266 ) from err
268 self._text = (parsed.content or "").strip()
270 self._date = parsed.created
271 if self._date is not None and timezone.is_naive(self._date):
272 self._date = timezone.make_aware(self._date)
274 # Always convert — requires_pdf_rendition=True means the browser
275 # cannot display the source format natively.
276 self._archive_path = self._convert_to_pdf(document_path)
278 # ------------------------------------------------------------------
279 # Result accessors
280 # ------------------------------------------------------------------
282 def get_text(self) -> str:
283 """Return the plain-text content extracted during parse.
285 Returns
286 -------
287 str
288 Extracted text, or an empty string if no text could be found.
289 """
290 return self._text or ""
292 def get_date(self) -> datetime.datetime | None:
293 """Return the document date detected during parse.
295 Returns
296 -------
297 datetime.datetime | None
298 Creation date from Tika metadata, or None if not detected.
299 """
300 return self._date
302 def get_archive_path(self) -> Path | None:
303 """Return the path to the generated PDF rendition, or None.
305 Returns
306 -------
307 Path | None
308 Path to the PDF produced by Gotenberg, or None if parse has not
309 been called yet.
310 """
311 return self._archive_path
313 # ------------------------------------------------------------------
314 # Thumbnail and metadata
315 # ------------------------------------------------------------------
317 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path:
318 """Generate a thumbnail from the PDF rendition of the document.
320 Converts the document to PDF first if not already done.
322 Parameters
323 ----------
324 document_path:
325 Absolute path to the source document.
326 mime_type:
327 Detected MIME type of the document.
329 Returns
330 -------
331 Path
332 Path to the generated WebP thumbnail inside the temporary directory.
333 """
334 if self._archive_path is None:
335 self._archive_path = self._convert_to_pdf(document_path)
336 return make_thumbnail_from_pdf(self._archive_path, self._tempdir)
338 def get_page_count(
339 self,
340 document_path: Path,
341 mime_type: str,
342 ) -> int | None:
343 """Return the number of pages in the document.
345 Counts pages in the archive PDF produced by a preceding parse()
346 call. Returns ``None`` if parse() has not been called yet or if
347 no archive was produced.
349 Returns
350 -------
351 int | None
352 Page count of the archive PDF, or ``None``.
353 """
354 if self._archive_path is not None:
355 from paperless.parsers.utils import get_page_count_for_pdf
357 return get_page_count_for_pdf(self._archive_path, log=logger)
358 return None
360 def extract_metadata(
361 self,
362 document_path: Path,
363 mime_type: str,
364 ) -> list[MetadataEntry]:
365 """Extract format-specific metadata via the Tika metadata endpoint.
367 Returns
368 -------
369 list[MetadataEntry]
370 All key/value pairs returned by Tika, or ``[]`` on error.
371 """
372 if TYPE_CHECKING:
373 assert self._tika_client is not None
375 try:
376 parsed = self._tika_client.metadata.from_file(document_path, mime_type)
377 return [
378 {
379 "namespace": "",
380 "prefix": "",
381 "key": key,
382 "value": parsed.data[key],
383 }
384 for key in parsed.data
385 ]
386 except Exception as e:
387 logger.warning(
388 "Error while fetching document metadata for %s: %s",
389 document_path,
390 e,
391 )
392 return []
394 # ------------------------------------------------------------------
395 # Private helpers
396 # ------------------------------------------------------------------
398 def _convert_to_pdf(self, document_path: Path) -> Path:
399 """Convert the document to PDF using Gotenberg's LibreOffice route.
401 Parameters
402 ----------
403 document_path:
404 Absolute path to the source document.
406 Returns
407 -------
408 Path
409 Path to the generated PDF inside the temporary directory.
411 Raises
412 ------
413 documents.parsers.ParseError
414 If Gotenberg returns an error.
415 """
416 if TYPE_CHECKING:
417 assert self._gotenberg_client is not None
419 pdf_path = self._tempdir / "convert.pdf"
421 logger.info("Converting %s to PDF as %s", document_path, pdf_path)
423 with self._gotenberg_client.libre_office.to_pdf() as route:
424 # Preserve document fields as authored. updateIndexes (Gotenberg's
425 # default) triggers a refresh() that rewrites dynamic fields like
426 # auto-dates to the current date.
427 route.update_indexes(update_indexes=False)
429 # Set the output format of the resulting PDF.
430 # OutputTypeConfig reads the database-stored ApplicationConfiguration
431 # first, then falls back to the PAPERLESS_OCR_OUTPUT_TYPE env var.
432 output_type = OutputTypeConfig().output_type
433 if output_type in {
434 OutputTypeChoices.PDF_A,
435 OutputTypeChoices.PDF_A2,
436 }:
437 route.pdf_format(PdfAFormat.A2b)
438 elif output_type == OutputTypeChoices.PDF_A1:
439 logger.warning(
440 "Gotenberg does not support PDF/A-1a, choosing PDF/A-2b instead",
441 )
442 route.pdf_format(PdfAFormat.A2b)
443 elif output_type == OutputTypeChoices.PDF_A3:
444 route.pdf_format(PdfAFormat.A3b)
446 route.convert(document_path)
448 try:
449 response = route.run()
450 pdf_path.write_bytes(response.content)
451 return pdf_path
452 except Exception as err:
453 raise ParseError(
454 f"Error while converting document to PDF: {err}",
455 ) from err