Coverage for paperless/parsers/tesseract.py: 19%
310 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import importlib.resources
4import logging
5import os
6import shutil
7import tempfile
8from pathlib import Path
9from typing import TYPE_CHECKING
10from typing import Any
11from typing import Final
12from typing import NoReturn
13from typing import Self
15from django.conf import settings
16from PIL import Image
18from documents.parsers import ParseError
19from documents.parsers import make_thumbnail_from_pdf
20from documents.utils import copy_file_with_basic_stats
21from documents.utils import maybe_override_pixel_limit
22from documents.utils import run_subprocess
23from paperless.config import OcrConfig
24from paperless.models import CleanChoices
25from paperless.models import ModeChoices
26from paperless.models import OutputTypeChoices
27from paperless.parsers.utils import extract_pdf_text
28from paperless.parsers.utils import is_born_digital_text
29from paperless.parsers.utils import post_process_text
30from paperless.parsers.utils import read_file_handle_unicode_errors
31from paperless.version import __full_version_str__
33if TYPE_CHECKING: 33 ↛ 34line 33 didn't jump to line 34 because the condition on line 33 was never true
34 import datetime
35 from types import TracebackType
37 from paperless.parsers import MetadataEntry
38 from paperless.parsers import ParserContext
40logger = logging.getLogger("paperless.parsing.tesseract")
42_SRGB_ICC_DATA: Final[bytes] = (
43 importlib.resources.files("ocrmypdf.data").joinpath("sRGB.icc").read_bytes()
44)
46_SUPPORTED_MIME_TYPES: Final[dict[str, str]] = {
47 "application/pdf": ".pdf",
48 "image/jpeg": ".jpg",
49 "image/png": ".png",
50 "image/tiff": ".tif",
51 "image/gif": ".gif",
52 "image/bmp": ".bmp",
53 "image/webp": ".webp",
54 "image/heic": ".heic",
55}
58class NoTextFoundException(Exception):
59 pass
62class RtlLanguageException(Exception):
63 pass
66class RasterisedDocumentParser:
67 """
68 This parser uses Tesseract to try and get some text out of a rasterised
69 image, whether it's a PDF, or other graphical format (JPEG, TIFF, etc.)
70 """
72 name: str = "Paperless-ngx Tesseract OCR Parser"
73 version: str = __full_version_str__
74 author: str = "Paperless-ngx Contributors"
75 url: str = "https://github.com/paperless-ngx/paperless-ngx"
77 # ------------------------------------------------------------------
78 # Class methods
79 # ------------------------------------------------------------------
81 @classmethod
82 def supported_mime_types(cls) -> dict[str, str]:
83 return _SUPPORTED_MIME_TYPES
85 @classmethod
86 def score(
87 cls,
88 mime_type: str,
89 filename: str,
90 path: Path | None = None,
91 ) -> int | None:
92 if mime_type in _SUPPORTED_MIME_TYPES: 92 ↛ 94line 92 didn't jump to line 94 because the condition on line 92 was always true
93 return 10
94 return None
96 # ------------------------------------------------------------------
97 # Properties
98 # ------------------------------------------------------------------
100 @property
101 def can_produce_archive(self) -> bool:
102 return True
104 @property
105 def requires_pdf_rendition(self) -> bool:
106 return False
108 # ------------------------------------------------------------------
109 # Lifecycle
110 # ------------------------------------------------------------------
112 def __init__(self, logging_group: object | None = None) -> None:
113 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True)
114 self.tempdir = Path(
115 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR),
116 )
117 self.settings = OcrConfig()
118 self.archive_path: Path | None = None
119 self.text: str | None = None
120 self.date: datetime.datetime | None = None
121 self.log = logger
123 def __enter__(self) -> Self:
124 return self
126 def __exit__(
127 self,
128 exc_type: type[BaseException] | None,
129 exc_val: BaseException | None,
130 exc_tb: TracebackType | None,
131 ) -> None:
132 logger.debug("Cleaning up temporary directory %s", self.tempdir)
133 shutil.rmtree(self.tempdir, ignore_errors=True)
135 # ------------------------------------------------------------------
136 # Core parsing interface
137 # ------------------------------------------------------------------
139 def configure(self, context: ParserContext) -> None:
140 pass
142 # ------------------------------------------------------------------
143 # Result accessors
144 # ------------------------------------------------------------------
146 def get_text(self) -> str:
147 return self.text or ""
149 def get_date(self) -> datetime.datetime | None:
150 return self.date
152 def get_archive_path(self) -> Path | None:
153 return self.archive_path
155 # ------------------------------------------------------------------
156 # Thumbnail, page count, and metadata
157 # ------------------------------------------------------------------
159 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path:
160 return make_thumbnail_from_pdf(
161 self.archive_path or Path(document_path),
162 self.tempdir,
163 )
165 def get_page_count(self, document_path: Path, mime_type: str) -> int | None:
166 if mime_type == "application/pdf":
167 from paperless.parsers.utils import get_page_count_for_pdf
169 return get_page_count_for_pdf(Path(document_path), log=self.log)
170 return None
172 def extract_metadata(
173 self,
174 document_path: Path,
175 mime_type: str,
176 ) -> list[MetadataEntry]:
177 if mime_type != "application/pdf":
178 return []
180 from paperless.parsers.utils import extract_pdf_metadata
182 return extract_pdf_metadata(Path(document_path), log=self.log)
184 def is_image(self, mime_type: str) -> bool:
185 return mime_type in [
186 "image/png",
187 "image/jpeg",
188 "image/tiff",
189 "image/bmp",
190 "image/gif",
191 "image/webp",
192 "image/heic",
193 ]
195 def has_alpha(self, image: Path) -> bool:
196 with Image.open(image) as im:
197 return im.mode in ("RGBA", "LA")
199 def remove_alpha(self, image_path: Path) -> Path:
200 no_alpha_image = Path(self.tempdir) / "image-no-alpha"
201 run_subprocess(
202 [
203 settings.CONVERT_BINARY,
204 "-alpha",
205 "off",
206 str(image_path),
207 str(no_alpha_image),
208 ],
209 logger=self.log,
210 )
211 return no_alpha_image
213 def get_dpi(self, image: Path) -> int | None:
214 try:
215 with Image.open(image) as im:
216 x, _ = im.info["dpi"]
217 return round(x)
218 except Exception as e:
219 self.log.warning(f"Error while getting DPI from image {image}: {e}")
220 return None
222 def calculate_a4_dpi(self, image: Path) -> int | None:
223 try:
224 with Image.open(image) as im:
225 width, _ = im.size
226 # divide image width by A4 width (210mm) in inches.
227 dpi = int(width / (21 / 2.54))
228 self.log.debug(f"Estimated DPI {dpi} based on image width {width}")
229 return dpi
231 except Exception as e:
232 self.log.warning(f"Error while calculating DPI for image {image}: {e}")
233 return None
235 def extract_text(
236 self,
237 sidecar_file: Path | None,
238 pdf_file: Path,
239 ) -> str | None:
240 text: str | None = None
241 # When re-doing OCR, the sidecar contains ONLY the new text, not
242 # the whole text, so do not utilize it in that case
243 if (
244 sidecar_file is not None
245 and sidecar_file.is_file()
246 and self.settings.mode != ModeChoices.REDO
247 ):
248 text = read_file_handle_unicode_errors(sidecar_file)
250 if "[OCR skipped on page" not in text:
251 # This happens when there's already text in the input file.
252 # The sidecar file will only contain text for OCR'ed pages.
253 self.log.debug("Using text from sidecar file")
254 return post_process_text(text)
255 else:
256 self.log.debug("Incomplete sidecar file: discarding.")
258 # no success with the sidecar file, try PDF
260 if not Path(pdf_file).is_file():
261 return None
263 return post_process_text(extract_pdf_text(Path(pdf_file), log=self.log))
265 def construct_ocrmypdf_parameters(
266 self,
267 input_file: Path,
268 mime_type: str,
269 output_file: Path,
270 sidecar_file: Path,
271 *,
272 safe_fallback: bool = False,
273 skip_text: bool = False,
274 ) -> dict[str, Any]:
275 ocrmypdf_args: dict[str, Any] = {
276 "input_file_or_options": input_file,
277 "output_file": output_file,
278 # need to use threads, since this will be run in daemonized
279 # processes via the task library.
280 "use_threads": True,
281 "jobs": settings.THREADS_PER_WORKER,
282 "language": self.settings.language,
283 "output_type": self.settings.output_type,
284 "progress_bar": False,
285 }
287 if "pdfa" in ocrmypdf_args["output_type"]:
288 ocrmypdf_args["color_conversion_strategy"] = (
289 self.settings.color_conversion_strategy
290 )
292 if safe_fallback or self.settings.mode == ModeChoices.FORCE:
293 ocrmypdf_args["force_ocr"] = True
294 elif self.settings.mode == ModeChoices.REDO:
295 ocrmypdf_args["redo_ocr"] = True
296 elif skip_text or self.settings.mode == ModeChoices.OFF:
297 ocrmypdf_args["skip_text"] = True
298 elif self.settings.mode == ModeChoices.AUTO:
299 pass # no extra flag: normal OCR (text not found case)
300 else: # pragma: no cover
301 raise ParseError(f"Invalid ocr mode: {self.settings.mode}")
303 if self.settings.clean == CleanChoices.CLEAN:
304 ocrmypdf_args["clean"] = True
305 elif self.settings.clean == CleanChoices.FINAL:
306 if self.settings.mode == ModeChoices.REDO:
307 ocrmypdf_args["clean"] = True
308 else:
309 # --clean-final is not compatible with --redo-ocr
310 ocrmypdf_args["clean_final"] = True
312 if self.settings.deskew and self.settings.mode != ModeChoices.REDO:
313 # --deskew is not compatible with --redo-ocr
314 ocrmypdf_args["deskew"] = True
316 if self.settings.rotate:
317 ocrmypdf_args["rotate_pages"] = True
318 ocrmypdf_args["rotate_pages_threshold"] = self.settings.rotate_threshold
320 if self.settings.pages is not None and self.settings.pages > 0:
321 ocrmypdf_args["pages"] = f"1-{self.settings.pages}"
322 else:
323 # sidecar is incompatible with pages
324 ocrmypdf_args["sidecar"] = sidecar_file
326 if self.is_image(mime_type):
327 # This may be required, depending on the known information
328 maybe_override_pixel_limit()
330 dpi = self.get_dpi(input_file)
331 a4_dpi = self.calculate_a4_dpi(input_file)
333 if self.has_alpha(input_file):
334 self.log.info(
335 f"Removing alpha layer from {input_file} "
336 "for compatibility with img2pdf",
337 )
338 # Replace the input file with the non-alpha
339 ocrmypdf_args["input_file_or_options"] = self.remove_alpha(input_file)
341 if dpi:
342 self.log.debug(f"Detected DPI for image {input_file}: {dpi}")
343 ocrmypdf_args["image_dpi"] = dpi
344 elif self.settings.image_dpi is not None:
345 ocrmypdf_args["image_dpi"] = self.settings.image_dpi
346 elif a4_dpi:
347 ocrmypdf_args["image_dpi"] = a4_dpi
348 else:
349 raise ParseError(
350 f"Cannot produce archive PDF for image {input_file}, "
351 f"no DPI information is present in this image and "
352 f"OCR_IMAGE_DPI is not set.",
353 )
354 if ocrmypdf_args["image_dpi"] < 70: # pragma: no cover
355 self.log.warning(
356 f"Image DPI of {ocrmypdf_args['image_dpi']} is low, OCR may fail",
357 )
359 if self.settings.user_args is not None:
360 try:
361 ocrmypdf_args = {**ocrmypdf_args, **self.settings.user_args}
362 except Exception as e:
363 self.log.warning(
364 f"There is an issue with PAPERLESS_OCR_USER_ARGS, so "
365 f"they will not be used. Error: {e}",
366 )
368 if (
369 self.settings.max_image_pixel is not None
370 and self.settings.max_image_pixel >= 0
371 ):
372 # Convert pixels to mega-pixels and provide to ocrmypdf
373 max_pixels_mpixels = self.settings.max_image_pixel / 1_000_000.0
374 msg = (
375 "OCR pixel limit is disabled!"
376 if max_pixels_mpixels == 0
377 else f"Calculated {max_pixels_mpixels} megapixels for OCR"
378 )
379 self.log.debug(msg)
380 ocrmypdf_args["max_image_mpixels"] = max_pixels_mpixels
382 return ocrmypdf_args
384 def _convert_image_to_pdfa(self, document_path: Path) -> Path:
385 """Convert an image to a PDF/A-2b file without invoking the OCR engine.
387 Uses img2pdf for the initial image->PDF wrapping, then pikepdf to stamp
388 PDF/A-2b conformance metadata.
390 No Tesseract and no Ghostscript are invoked.
391 """
392 import img2pdf
393 import pikepdf
395 plain_pdf_path = Path(self.tempdir) / "image_plain.pdf"
396 try:
397 convert_kwargs: dict = {
398 # Ignore invalid EXIF orientation values (e.g. 0) instead of
399 # aborting the conversion; valid values are still applied
400 "rotation": img2pdf.Rotation.ifvalid,
401 }
402 if self.settings.image_dpi is not None:
403 convert_kwargs["layout_fun"] = img2pdf.get_fixed_dpi_layout_fun(
404 (self.settings.image_dpi, self.settings.image_dpi),
405 )
406 plain_pdf_path.write_bytes(
407 img2pdf.convert(str(document_path), **convert_kwargs),
408 )
409 except Exception as e:
410 raise ParseError(
411 f"img2pdf conversion failed for {document_path}: {e!s}",
412 ) from e
414 pdfa_path = Path(self.tempdir) / "archive.pdf"
415 try:
416 with pikepdf.open(plain_pdf_path) as pdf:
417 cs = pdf.make_stream(_SRGB_ICC_DATA)
418 cs["/N"] = 3
419 output_intent = pikepdf.Dictionary(
420 Type=pikepdf.Name("/OutputIntent"),
421 S=pikepdf.Name("/GTS_PDFA1"),
422 OutputConditionIdentifier=pikepdf.String("sRGB"),
423 DestOutputProfile=cs,
424 )
425 pdf.Root["/OutputIntents"] = pdf.make_indirect(
426 pikepdf.Array([output_intent]),
427 )
428 meta = pdf.open_metadata(set_pikepdf_as_editor=False)
429 meta["pdfaid:part"] = "2"
430 meta["pdfaid:conformance"] = "B"
431 pdf.save(pdfa_path)
432 except Exception as e:
433 self.log.warning(
434 f"PDF/A metadata stamping failed ({e!s}); falling back to plain PDF.",
435 )
436 pdfa_path.write_bytes(plain_pdf_path.read_bytes())
438 return pdfa_path
440 def _convert_pdf_to_pdfa(
441 self,
442 input_path: Path,
443 output_path: Path,
444 ) -> None:
445 """Convert a PDF to PDF/A using Ghostscript directly, without OCR.
447 Respects the user's output_type, color_conversion_strategy, and
448 continue_on_soft_render_error settings.
449 """
450 from ocrmypdf._exec.ghostscript import generate_pdfa
451 from ocrmypdf.pdfa import generate_pdfa_ps
453 output_type = self.settings.output_type
454 if output_type == OutputTypeChoices.PDF:
455 # No PDF/A requested — just copy the original
456 copy_file_with_basic_stats(input_path, output_path)
457 return
459 # Map output_type to pdfa_part: pdfa→2, pdfa-1→1, pdfa-2→2, pdfa-3→3
460 pdfa_part = "2" if output_type == "pdfa" else output_type.split("-")[-1]
462 pdfmark = Path(self.tempdir) / "pdfa.ps"
463 generate_pdfa_ps(pdfmark)
465 color_strategy = self.settings.color_conversion_strategy or "RGB"
467 self.log.debug(
468 "Converting PDF to PDF/A-%s via Ghostscript (no OCR): %s",
469 pdfa_part,
470 input_path,
471 )
473 generate_pdfa(
474 pdf_pages=[pdfmark, input_path],
475 output_file=output_path,
476 compression="auto",
477 color_conversion_strategy=color_strategy,
478 pdfa_part=pdfa_part,
479 )
481 def _handle_subprocess_output_error(self, e: Exception) -> NoReturn:
482 """Log context for Ghostscript failures and raise ParseError.
484 Called from the SubprocessOutputError handlers in parse() to avoid
485 duplicating the Ghostscript hint and re-raise logic.
486 """
487 if "Ghostscript PDF/A rendering" in str(e):
488 self.log.warning(
489 "Ghostscript PDF/A rendering failed, consider setting "
490 "PAPERLESS_OCR_USER_ARGS: "
491 "'{\"continue_on_soft_render_error\": true}'",
492 )
493 raise ParseError(
494 f"SubprocessOutputError: {e!s}. See logs for more information.",
495 ) from e
497 def parse(
498 self,
499 document_path: Path,
500 mime_type: str,
501 *,
502 produce_archive: bool = True,
503 ) -> None:
504 # This forces tesseract to use one core per page.
505 os.environ["OMP_THREAD_LIMIT"] = "1"
507 import ocrmypdf
508 from ocrmypdf import EncryptedPdfError
509 from ocrmypdf import InputFileError
510 from ocrmypdf import SubprocessOutputError
511 from ocrmypdf.exceptions import DigitalSignatureError
512 from ocrmypdf.exceptions import PriorOcrFoundError
514 if mime_type == "application/pdf":
515 text_original = self.extract_text(None, document_path)
516 original_has_text = is_born_digital_text(
517 text_original,
518 document_path,
519 log=self.log,
520 )
521 else:
522 text_original = None
523 original_has_text = False
525 self.log.debug(
526 "Text detection: original_has_text=%s (text_length=%d, mode=%s, produce_archive=%s)",
527 original_has_text,
528 len(text_original) if text_original else 0,
529 self.settings.mode,
530 produce_archive,
531 )
533 # --- OCR_MODE=off: never invoke OCR engine ---
534 if self.settings.mode == ModeChoices.OFF:
535 if not produce_archive:
536 self.log.debug(
537 "OCR: skipped — OCR_MODE=off, no archive requested;"
538 " returning pdftotext content only",
539 )
540 self.text = text_original or ""
541 return
542 if self.is_image(mime_type):
543 self.log.debug(
544 "OCR: skipped — OCR_MODE=off, image input;"
545 " converting to PDF/A without OCR",
546 )
547 try:
548 self.archive_path = self._convert_image_to_pdfa(
549 document_path,
550 )
551 self.text = ""
552 except Exception as e:
553 raise ParseError(
554 f"Image to PDF/A conversion failed: {e!s}",
555 ) from e
556 return
557 # PDFs in off mode: PDF/A conversion via Ghostscript, no OCR
558 archive_path = Path(self.tempdir) / "archive.pdf"
559 try:
560 self._convert_pdf_to_pdfa(document_path, archive_path)
561 self.archive_path = archive_path
562 self.text = text_original or ""
563 except SubprocessOutputError as e:
564 self._handle_subprocess_output_error(e)
565 except Exception as e:
566 raise ParseError(f"{e.__class__.__name__}: {e!s}") from e
567 return
569 # --- OCR_MODE=auto: skip ocrmypdf entirely if text exists and no archive needed ---
570 if (
571 self.settings.mode == ModeChoices.AUTO
572 and original_has_text
573 and not produce_archive
574 ):
575 self.log.debug(
576 "Document has text and no archive requested; skipping OCRmyPDF entirely.",
577 )
578 self.text = text_original
579 return
581 # --- All other paths: run ocrmypdf ---
582 archive_path = Path(self.tempdir) / "archive.pdf"
583 sidecar_file = Path(self.tempdir) / "sidecar.txt"
585 # auto mode with existing text: PDF/A conversion only (no OCR).
586 skip_text = self.settings.mode == ModeChoices.AUTO and original_has_text
588 if skip_text:
589 self.log.debug(
590 "OCR strategy: PDF/A conversion only (skip_text)"
591 " — OCR_MODE=auto, document already has text",
592 )
593 else:
594 self.log.debug("OCR strategy: full OCR — OCR_MODE=%s", self.settings.mode)
596 args = self.construct_ocrmypdf_parameters(
597 document_path,
598 mime_type,
599 archive_path,
600 sidecar_file,
601 skip_text=skip_text,
602 )
604 try:
605 self.log.debug(f"Calling OCRmyPDF with args: {args}")
606 ocrmypdf.ocr(**args)
608 if produce_archive:
609 self.archive_path = archive_path
611 self.text = self.extract_text(sidecar_file, archive_path)
613 if not self.text:
614 raise NoTextFoundException("No text was found in the original document")
615 except (DigitalSignatureError, EncryptedPdfError):
616 self.log.warning(
617 "This file is encrypted and/or signed, OCR is impossible. Using "
618 "any text present in the original file.",
619 )
620 if original_has_text:
621 self.text = text_original
622 except SubprocessOutputError as e:
623 self._handle_subprocess_output_error(e)
624 except (NoTextFoundException, InputFileError, PriorOcrFoundError) as e:
625 self.log.warning(
626 f"Encountered an error while running OCR: {e!s}. "
627 f"Attempting force OCR to get the text.",
628 )
630 archive_path_fallback = Path(self.tempdir) / "archive-fallback.pdf"
631 sidecar_file_fallback = Path(self.tempdir) / "sidecar-fallback.txt"
633 args = self.construct_ocrmypdf_parameters(
634 document_path,
635 mime_type,
636 archive_path_fallback,
637 sidecar_file_fallback,
638 safe_fallback=True,
639 )
641 try:
642 self.log.debug(f"Fallback: Calling OCRmyPDF with args: {args}")
643 ocrmypdf.ocr(**args)
644 self.text = self.extract_text(
645 sidecar_file_fallback,
646 archive_path_fallback,
647 )
648 if produce_archive:
649 self.archive_path = archive_path_fallback
650 except Exception as e:
651 raise ParseError(f"{e.__class__.__name__}: {e!s}") from e
653 except Exception as e:
654 raise ParseError(f"{e.__class__.__name__}: {e!s}") from e
656 if not self.text:
657 if original_has_text:
658 self.text = text_original
659 else:
660 self.log.warning(
661 f"No text was found in {document_path}, the content will be empty.",
662 )
663 self.text = ""