Coverage for paperless/parsers/tesseract.py: 19%

310 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import importlib.resources 

4import logging 

5import os 

6import shutil 

7import tempfile 

8from pathlib import Path 

9from typing import TYPE_CHECKING 

10from typing import Any 

11from typing import Final 

12from typing import NoReturn 

13from typing import Self 

14 

15from django.conf import settings 

16from PIL import Image 

17 

18from documents.parsers import ParseError 

19from documents.parsers import make_thumbnail_from_pdf 

20from documents.utils import copy_file_with_basic_stats 

21from documents.utils import maybe_override_pixel_limit 

22from documents.utils import run_subprocess 

23from paperless.config import OcrConfig 

24from paperless.models import CleanChoices 

25from paperless.models import ModeChoices 

26from paperless.models import OutputTypeChoices 

27from paperless.parsers.utils import extract_pdf_text 

28from paperless.parsers.utils import is_born_digital_text 

29from paperless.parsers.utils import post_process_text 

30from paperless.parsers.utils import read_file_handle_unicode_errors 

31from paperless.version import __full_version_str__ 

32 

33if TYPE_CHECKING: 33 ↛ 34line 33 didn't jump to line 34 because the condition on line 33 was never true

34 import datetime 

35 from types import TracebackType 

36 

37 from paperless.parsers import MetadataEntry 

38 from paperless.parsers import ParserContext 

39 

40logger = logging.getLogger("paperless.parsing.tesseract") 

41 

42_SRGB_ICC_DATA: Final[bytes] = ( 

43 importlib.resources.files("ocrmypdf.data").joinpath("sRGB.icc").read_bytes() 

44) 

45 

46_SUPPORTED_MIME_TYPES: Final[dict[str, str]] = { 

47 "application/pdf": ".pdf", 

48 "image/jpeg": ".jpg", 

49 "image/png": ".png", 

50 "image/tiff": ".tif", 

51 "image/gif": ".gif", 

52 "image/bmp": ".bmp", 

53 "image/webp": ".webp", 

54 "image/heic": ".heic", 

55} 

56 

57 

58class NoTextFoundException(Exception): 

59 pass 

60 

61 

62class RtlLanguageException(Exception): 

63 pass 

64 

65 

66class RasterisedDocumentParser: 

67 """ 

68 This parser uses Tesseract to try and get some text out of a rasterised 

69 image, whether it's a PDF, or other graphical format (JPEG, TIFF, etc.) 

70 """ 

71 

72 name: str = "Paperless-ngx Tesseract OCR Parser" 

73 version: str = __full_version_str__ 

74 author: str = "Paperless-ngx Contributors" 

75 url: str = "https://github.com/paperless-ngx/paperless-ngx" 

76 

77 # ------------------------------------------------------------------ 

78 # Class methods 

79 # ------------------------------------------------------------------ 

80 

81 @classmethod 

82 def supported_mime_types(cls) -> dict[str, str]: 

83 return _SUPPORTED_MIME_TYPES 

84 

85 @classmethod 

86 def score( 

87 cls, 

88 mime_type: str, 

89 filename: str, 

90 path: Path | None = None, 

91 ) -> int | None: 

92 if mime_type in _SUPPORTED_MIME_TYPES: 92 ↛ 94line 92 didn't jump to line 94 because the condition on line 92 was always true

93 return 10 

94 return None 

95 

96 # ------------------------------------------------------------------ 

97 # Properties 

98 # ------------------------------------------------------------------ 

99 

100 @property 

101 def can_produce_archive(self) -> bool: 

102 return True 

103 

104 @property 

105 def requires_pdf_rendition(self) -> bool: 

106 return False 

107 

108 # ------------------------------------------------------------------ 

109 # Lifecycle 

110 # ------------------------------------------------------------------ 

111 

112 def __init__(self, logging_group: object | None = None) -> None: 

113 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True) 

114 self.tempdir = Path( 

115 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR), 

116 ) 

117 self.settings = OcrConfig() 

118 self.archive_path: Path | None = None 

119 self.text: str | None = None 

120 self.date: datetime.datetime | None = None 

121 self.log = logger 

122 

123 def __enter__(self) -> Self: 

124 return self 

125 

126 def __exit__( 

127 self, 

128 exc_type: type[BaseException] | None, 

129 exc_val: BaseException | None, 

130 exc_tb: TracebackType | None, 

131 ) -> None: 

132 logger.debug("Cleaning up temporary directory %s", self.tempdir) 

133 shutil.rmtree(self.tempdir, ignore_errors=True) 

134 

135 # ------------------------------------------------------------------ 

136 # Core parsing interface 

137 # ------------------------------------------------------------------ 

138 

139 def configure(self, context: ParserContext) -> None: 

140 pass 

141 

142 # ------------------------------------------------------------------ 

143 # Result accessors 

144 # ------------------------------------------------------------------ 

145 

146 def get_text(self) -> str: 

147 return self.text or "" 

148 

149 def get_date(self) -> datetime.datetime | None: 

150 return self.date 

151 

152 def get_archive_path(self) -> Path | None: 

153 return self.archive_path 

154 

155 # ------------------------------------------------------------------ 

156 # Thumbnail, page count, and metadata 

157 # ------------------------------------------------------------------ 

158 

159 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path: 

160 return make_thumbnail_from_pdf( 

161 self.archive_path or Path(document_path), 

162 self.tempdir, 

163 ) 

164 

165 def get_page_count(self, document_path: Path, mime_type: str) -> int | None: 

166 if mime_type == "application/pdf": 

167 from paperless.parsers.utils import get_page_count_for_pdf 

168 

169 return get_page_count_for_pdf(Path(document_path), log=self.log) 

170 return None 

171 

172 def extract_metadata( 

173 self, 

174 document_path: Path, 

175 mime_type: str, 

176 ) -> list[MetadataEntry]: 

177 if mime_type != "application/pdf": 

178 return [] 

179 

180 from paperless.parsers.utils import extract_pdf_metadata 

181 

182 return extract_pdf_metadata(Path(document_path), log=self.log) 

183 

184 def is_image(self, mime_type: str) -> bool: 

185 return mime_type in [ 

186 "image/png", 

187 "image/jpeg", 

188 "image/tiff", 

189 "image/bmp", 

190 "image/gif", 

191 "image/webp", 

192 "image/heic", 

193 ] 

194 

195 def has_alpha(self, image: Path) -> bool: 

196 with Image.open(image) as im: 

197 return im.mode in ("RGBA", "LA") 

198 

199 def remove_alpha(self, image_path: Path) -> Path: 

200 no_alpha_image = Path(self.tempdir) / "image-no-alpha" 

201 run_subprocess( 

202 [ 

203 settings.CONVERT_BINARY, 

204 "-alpha", 

205 "off", 

206 str(image_path), 

207 str(no_alpha_image), 

208 ], 

209 logger=self.log, 

210 ) 

211 return no_alpha_image 

212 

213 def get_dpi(self, image: Path) -> int | None: 

214 try: 

215 with Image.open(image) as im: 

216 x, _ = im.info["dpi"] 

217 return round(x) 

218 except Exception as e: 

219 self.log.warning(f"Error while getting DPI from image {image}: {e}") 

220 return None 

221 

222 def calculate_a4_dpi(self, image: Path) -> int | None: 

223 try: 

224 with Image.open(image) as im: 

225 width, _ = im.size 

226 # divide image width by A4 width (210mm) in inches. 

227 dpi = int(width / (21 / 2.54)) 

228 self.log.debug(f"Estimated DPI {dpi} based on image width {width}") 

229 return dpi 

230 

231 except Exception as e: 

232 self.log.warning(f"Error while calculating DPI for image {image}: {e}") 

233 return None 

234 

235 def extract_text( 

236 self, 

237 sidecar_file: Path | None, 

238 pdf_file: Path, 

239 ) -> str | None: 

240 text: str | None = None 

241 # When re-doing OCR, the sidecar contains ONLY the new text, not 

242 # the whole text, so do not utilize it in that case 

243 if ( 

244 sidecar_file is not None 

245 and sidecar_file.is_file() 

246 and self.settings.mode != ModeChoices.REDO 

247 ): 

248 text = read_file_handle_unicode_errors(sidecar_file) 

249 

250 if "[OCR skipped on page" not in text: 

251 # This happens when there's already text in the input file. 

252 # The sidecar file will only contain text for OCR'ed pages. 

253 self.log.debug("Using text from sidecar file") 

254 return post_process_text(text) 

255 else: 

256 self.log.debug("Incomplete sidecar file: discarding.") 

257 

258 # no success with the sidecar file, try PDF 

259 

260 if not Path(pdf_file).is_file(): 

261 return None 

262 

263 return post_process_text(extract_pdf_text(Path(pdf_file), log=self.log)) 

264 

265 def construct_ocrmypdf_parameters( 

266 self, 

267 input_file: Path, 

268 mime_type: str, 

269 output_file: Path, 

270 sidecar_file: Path, 

271 *, 

272 safe_fallback: bool = False, 

273 skip_text: bool = False, 

274 ) -> dict[str, Any]: 

275 ocrmypdf_args: dict[str, Any] = { 

276 "input_file_or_options": input_file, 

277 "output_file": output_file, 

278 # need to use threads, since this will be run in daemonized 

279 # processes via the task library. 

280 "use_threads": True, 

281 "jobs": settings.THREADS_PER_WORKER, 

282 "language": self.settings.language, 

283 "output_type": self.settings.output_type, 

284 "progress_bar": False, 

285 } 

286 

287 if "pdfa" in ocrmypdf_args["output_type"]: 

288 ocrmypdf_args["color_conversion_strategy"] = ( 

289 self.settings.color_conversion_strategy 

290 ) 

291 

292 if safe_fallback or self.settings.mode == ModeChoices.FORCE: 

293 ocrmypdf_args["force_ocr"] = True 

294 elif self.settings.mode == ModeChoices.REDO: 

295 ocrmypdf_args["redo_ocr"] = True 

296 elif skip_text or self.settings.mode == ModeChoices.OFF: 

297 ocrmypdf_args["skip_text"] = True 

298 elif self.settings.mode == ModeChoices.AUTO: 

299 pass # no extra flag: normal OCR (text not found case) 

300 else: # pragma: no cover 

301 raise ParseError(f"Invalid ocr mode: {self.settings.mode}") 

302 

303 if self.settings.clean == CleanChoices.CLEAN: 

304 ocrmypdf_args["clean"] = True 

305 elif self.settings.clean == CleanChoices.FINAL: 

306 if self.settings.mode == ModeChoices.REDO: 

307 ocrmypdf_args["clean"] = True 

308 else: 

309 # --clean-final is not compatible with --redo-ocr 

310 ocrmypdf_args["clean_final"] = True 

311 

312 if self.settings.deskew and self.settings.mode != ModeChoices.REDO: 

313 # --deskew is not compatible with --redo-ocr 

314 ocrmypdf_args["deskew"] = True 

315 

316 if self.settings.rotate: 

317 ocrmypdf_args["rotate_pages"] = True 

318 ocrmypdf_args["rotate_pages_threshold"] = self.settings.rotate_threshold 

319 

320 if self.settings.pages is not None and self.settings.pages > 0: 

321 ocrmypdf_args["pages"] = f"1-{self.settings.pages}" 

322 else: 

323 # sidecar is incompatible with pages 

324 ocrmypdf_args["sidecar"] = sidecar_file 

325 

326 if self.is_image(mime_type): 

327 # This may be required, depending on the known information 

328 maybe_override_pixel_limit() 

329 

330 dpi = self.get_dpi(input_file) 

331 a4_dpi = self.calculate_a4_dpi(input_file) 

332 

333 if self.has_alpha(input_file): 

334 self.log.info( 

335 f"Removing alpha layer from {input_file} " 

336 "for compatibility with img2pdf", 

337 ) 

338 # Replace the input file with the non-alpha 

339 ocrmypdf_args["input_file_or_options"] = self.remove_alpha(input_file) 

340 

341 if dpi: 

342 self.log.debug(f"Detected DPI for image {input_file}: {dpi}") 

343 ocrmypdf_args["image_dpi"] = dpi 

344 elif self.settings.image_dpi is not None: 

345 ocrmypdf_args["image_dpi"] = self.settings.image_dpi 

346 elif a4_dpi: 

347 ocrmypdf_args["image_dpi"] = a4_dpi 

348 else: 

349 raise ParseError( 

350 f"Cannot produce archive PDF for image {input_file}, " 

351 f"no DPI information is present in this image and " 

352 f"OCR_IMAGE_DPI is not set.", 

353 ) 

354 if ocrmypdf_args["image_dpi"] < 70: # pragma: no cover 

355 self.log.warning( 

356 f"Image DPI of {ocrmypdf_args['image_dpi']} is low, OCR may fail", 

357 ) 

358 

359 if self.settings.user_args is not None: 

360 try: 

361 ocrmypdf_args = {**ocrmypdf_args, **self.settings.user_args} 

362 except Exception as e: 

363 self.log.warning( 

364 f"There is an issue with PAPERLESS_OCR_USER_ARGS, so " 

365 f"they will not be used. Error: {e}", 

366 ) 

367 

368 if ( 

369 self.settings.max_image_pixel is not None 

370 and self.settings.max_image_pixel >= 0 

371 ): 

372 # Convert pixels to mega-pixels and provide to ocrmypdf 

373 max_pixels_mpixels = self.settings.max_image_pixel / 1_000_000.0 

374 msg = ( 

375 "OCR pixel limit is disabled!" 

376 if max_pixels_mpixels == 0 

377 else f"Calculated {max_pixels_mpixels} megapixels for OCR" 

378 ) 

379 self.log.debug(msg) 

380 ocrmypdf_args["max_image_mpixels"] = max_pixels_mpixels 

381 

382 return ocrmypdf_args 

383 

384 def _convert_image_to_pdfa(self, document_path: Path) -> Path: 

385 """Convert an image to a PDF/A-2b file without invoking the OCR engine. 

386 

387 Uses img2pdf for the initial image->PDF wrapping, then pikepdf to stamp 

388 PDF/A-2b conformance metadata. 

389 

390 No Tesseract and no Ghostscript are invoked. 

391 """ 

392 import img2pdf 

393 import pikepdf 

394 

395 plain_pdf_path = Path(self.tempdir) / "image_plain.pdf" 

396 try: 

397 convert_kwargs: dict = { 

398 # Ignore invalid EXIF orientation values (e.g. 0) instead of 

399 # aborting the conversion; valid values are still applied 

400 "rotation": img2pdf.Rotation.ifvalid, 

401 } 

402 if self.settings.image_dpi is not None: 

403 convert_kwargs["layout_fun"] = img2pdf.get_fixed_dpi_layout_fun( 

404 (self.settings.image_dpi, self.settings.image_dpi), 

405 ) 

406 plain_pdf_path.write_bytes( 

407 img2pdf.convert(str(document_path), **convert_kwargs), 

408 ) 

409 except Exception as e: 

410 raise ParseError( 

411 f"img2pdf conversion failed for {document_path}: {e!s}", 

412 ) from e 

413 

414 pdfa_path = Path(self.tempdir) / "archive.pdf" 

415 try: 

416 with pikepdf.open(plain_pdf_path) as pdf: 

417 cs = pdf.make_stream(_SRGB_ICC_DATA) 

418 cs["/N"] = 3 

419 output_intent = pikepdf.Dictionary( 

420 Type=pikepdf.Name("/OutputIntent"), 

421 S=pikepdf.Name("/GTS_PDFA1"), 

422 OutputConditionIdentifier=pikepdf.String("sRGB"), 

423 DestOutputProfile=cs, 

424 ) 

425 pdf.Root["/OutputIntents"] = pdf.make_indirect( 

426 pikepdf.Array([output_intent]), 

427 ) 

428 meta = pdf.open_metadata(set_pikepdf_as_editor=False) 

429 meta["pdfaid:part"] = "2" 

430 meta["pdfaid:conformance"] = "B" 

431 pdf.save(pdfa_path) 

432 except Exception as e: 

433 self.log.warning( 

434 f"PDF/A metadata stamping failed ({e!s}); falling back to plain PDF.", 

435 ) 

436 pdfa_path.write_bytes(plain_pdf_path.read_bytes()) 

437 

438 return pdfa_path 

439 

440 def _convert_pdf_to_pdfa( 

441 self, 

442 input_path: Path, 

443 output_path: Path, 

444 ) -> None: 

445 """Convert a PDF to PDF/A using Ghostscript directly, without OCR. 

446 

447 Respects the user's output_type, color_conversion_strategy, and 

448 continue_on_soft_render_error settings. 

449 """ 

450 from ocrmypdf._exec.ghostscript import generate_pdfa 

451 from ocrmypdf.pdfa import generate_pdfa_ps 

452 

453 output_type = self.settings.output_type 

454 if output_type == OutputTypeChoices.PDF: 

455 # No PDF/A requested — just copy the original 

456 copy_file_with_basic_stats(input_path, output_path) 

457 return 

458 

459 # Map output_type to pdfa_part: pdfa→2, pdfa-1→1, pdfa-2→2, pdfa-3→3 

460 pdfa_part = "2" if output_type == "pdfa" else output_type.split("-")[-1] 

461 

462 pdfmark = Path(self.tempdir) / "pdfa.ps" 

463 generate_pdfa_ps(pdfmark) 

464 

465 color_strategy = self.settings.color_conversion_strategy or "RGB" 

466 

467 self.log.debug( 

468 "Converting PDF to PDF/A-%s via Ghostscript (no OCR): %s", 

469 pdfa_part, 

470 input_path, 

471 ) 

472 

473 generate_pdfa( 

474 pdf_pages=[pdfmark, input_path], 

475 output_file=output_path, 

476 compression="auto", 

477 color_conversion_strategy=color_strategy, 

478 pdfa_part=pdfa_part, 

479 ) 

480 

481 def _handle_subprocess_output_error(self, e: Exception) -> NoReturn: 

482 """Log context for Ghostscript failures and raise ParseError. 

483 

484 Called from the SubprocessOutputError handlers in parse() to avoid 

485 duplicating the Ghostscript hint and re-raise logic. 

486 """ 

487 if "Ghostscript PDF/A rendering" in str(e): 

488 self.log.warning( 

489 "Ghostscript PDF/A rendering failed, consider setting " 

490 "PAPERLESS_OCR_USER_ARGS: " 

491 "'{\"continue_on_soft_render_error\": true}'", 

492 ) 

493 raise ParseError( 

494 f"SubprocessOutputError: {e!s}. See logs for more information.", 

495 ) from e 

496 

497 def parse( 

498 self, 

499 document_path: Path, 

500 mime_type: str, 

501 *, 

502 produce_archive: bool = True, 

503 ) -> None: 

504 # This forces tesseract to use one core per page. 

505 os.environ["OMP_THREAD_LIMIT"] = "1" 

506 

507 import ocrmypdf 

508 from ocrmypdf import EncryptedPdfError 

509 from ocrmypdf import InputFileError 

510 from ocrmypdf import SubprocessOutputError 

511 from ocrmypdf.exceptions import DigitalSignatureError 

512 from ocrmypdf.exceptions import PriorOcrFoundError 

513 

514 if mime_type == "application/pdf": 

515 text_original = self.extract_text(None, document_path) 

516 original_has_text = is_born_digital_text( 

517 text_original, 

518 document_path, 

519 log=self.log, 

520 ) 

521 else: 

522 text_original = None 

523 original_has_text = False 

524 

525 self.log.debug( 

526 "Text detection: original_has_text=%s (text_length=%d, mode=%s, produce_archive=%s)", 

527 original_has_text, 

528 len(text_original) if text_original else 0, 

529 self.settings.mode, 

530 produce_archive, 

531 ) 

532 

533 # --- OCR_MODE=off: never invoke OCR engine --- 

534 if self.settings.mode == ModeChoices.OFF: 

535 if not produce_archive: 

536 self.log.debug( 

537 "OCR: skipped — OCR_MODE=off, no archive requested;" 

538 " returning pdftotext content only", 

539 ) 

540 self.text = text_original or "" 

541 return 

542 if self.is_image(mime_type): 

543 self.log.debug( 

544 "OCR: skipped — OCR_MODE=off, image input;" 

545 " converting to PDF/A without OCR", 

546 ) 

547 try: 

548 self.archive_path = self._convert_image_to_pdfa( 

549 document_path, 

550 ) 

551 self.text = "" 

552 except Exception as e: 

553 raise ParseError( 

554 f"Image to PDF/A conversion failed: {e!s}", 

555 ) from e 

556 return 

557 # PDFs in off mode: PDF/A conversion via Ghostscript, no OCR 

558 archive_path = Path(self.tempdir) / "archive.pdf" 

559 try: 

560 self._convert_pdf_to_pdfa(document_path, archive_path) 

561 self.archive_path = archive_path 

562 self.text = text_original or "" 

563 except SubprocessOutputError as e: 

564 self._handle_subprocess_output_error(e) 

565 except Exception as e: 

566 raise ParseError(f"{e.__class__.__name__}: {e!s}") from e 

567 return 

568 

569 # --- OCR_MODE=auto: skip ocrmypdf entirely if text exists and no archive needed --- 

570 if ( 

571 self.settings.mode == ModeChoices.AUTO 

572 and original_has_text 

573 and not produce_archive 

574 ): 

575 self.log.debug( 

576 "Document has text and no archive requested; skipping OCRmyPDF entirely.", 

577 ) 

578 self.text = text_original 

579 return 

580 

581 # --- All other paths: run ocrmypdf --- 

582 archive_path = Path(self.tempdir) / "archive.pdf" 

583 sidecar_file = Path(self.tempdir) / "sidecar.txt" 

584 

585 # auto mode with existing text: PDF/A conversion only (no OCR). 

586 skip_text = self.settings.mode == ModeChoices.AUTO and original_has_text 

587 

588 if skip_text: 

589 self.log.debug( 

590 "OCR strategy: PDF/A conversion only (skip_text)" 

591 " — OCR_MODE=auto, document already has text", 

592 ) 

593 else: 

594 self.log.debug("OCR strategy: full OCR — OCR_MODE=%s", self.settings.mode) 

595 

596 args = self.construct_ocrmypdf_parameters( 

597 document_path, 

598 mime_type, 

599 archive_path, 

600 sidecar_file, 

601 skip_text=skip_text, 

602 ) 

603 

604 try: 

605 self.log.debug(f"Calling OCRmyPDF with args: {args}") 

606 ocrmypdf.ocr(**args) 

607 

608 if produce_archive: 

609 self.archive_path = archive_path 

610 

611 self.text = self.extract_text(sidecar_file, archive_path) 

612 

613 if not self.text: 

614 raise NoTextFoundException("No text was found in the original document") 

615 except (DigitalSignatureError, EncryptedPdfError): 

616 self.log.warning( 

617 "This file is encrypted and/or signed, OCR is impossible. Using " 

618 "any text present in the original file.", 

619 ) 

620 if original_has_text: 

621 self.text = text_original 

622 except SubprocessOutputError as e: 

623 self._handle_subprocess_output_error(e) 

624 except (NoTextFoundException, InputFileError, PriorOcrFoundError) as e: 

625 self.log.warning( 

626 f"Encountered an error while running OCR: {e!s}. " 

627 f"Attempting force OCR to get the text.", 

628 ) 

629 

630 archive_path_fallback = Path(self.tempdir) / "archive-fallback.pdf" 

631 sidecar_file_fallback = Path(self.tempdir) / "sidecar-fallback.txt" 

632 

633 args = self.construct_ocrmypdf_parameters( 

634 document_path, 

635 mime_type, 

636 archive_path_fallback, 

637 sidecar_file_fallback, 

638 safe_fallback=True, 

639 ) 

640 

641 try: 

642 self.log.debug(f"Fallback: Calling OCRmyPDF with args: {args}") 

643 ocrmypdf.ocr(**args) 

644 self.text = self.extract_text( 

645 sidecar_file_fallback, 

646 archive_path_fallback, 

647 ) 

648 if produce_archive: 

649 self.archive_path = archive_path_fallback 

650 except Exception as e: 

651 raise ParseError(f"{e.__class__.__name__}: {e!s}") from e 

652 

653 except Exception as e: 

654 raise ParseError(f"{e.__class__.__name__}: {e!s}") from e 

655 

656 if not self.text: 

657 if original_has_text: 

658 self.text = text_original 

659 else: 

660 self.log.warning( 

661 f"No text was found in {document_path}, the content will be empty.", 

662 ) 

663 self.text = ""