Coverage for paperless/parsers/utils.py: 16%
106 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1"""
2Shared utilities for Paperless-ngx document parsers.
4Functions here are format-neutral helpers that multiple parsers need.
5Keeping them here avoids parsers inheriting from each other just to
6share implementation.
7"""
9from __future__ import annotations
11import codecs
12import logging
13import re
14import tempfile
15from pathlib import Path
16from typing import TYPE_CHECKING
17from typing import Final
19if TYPE_CHECKING: 19 ↛ 20line 19 didn't jump to line 20 because the condition on line 19 was never true
20 from paperless.parsers import MetadataEntry
22logger = logging.getLogger("paperless.parsers.utils")
24# Minimum character count for a PDF to be considered "born-digital" (has real text).
25# Used by both the consumer (archive decision) and the tesseract parser (skip-OCR decision).
26PDF_TEXT_MIN_LENGTH: Final[int] = 50
29def is_tagged_pdf(
30 path: Path,
31 log: logging.Logger | None = None,
32) -> bool:
33 """Return True if the PDF declares itself as tagged (born-digital indicator).
35 Tagged PDFs (e.g. exported from Word or LibreOffice) have ``/MarkInfo``
36 with ``/Marked true`` in the document root. This is a reliable signal
37 that the document has a logical structure and embedded text — running OCR
38 on it is unnecessary and archive generation can be skipped.
40 https://github.com/ocrmypdf/OCRmyPDF/blob/4e974ebd465a5921b2e79004f098f5d203010282/src/ocrmypdf/pdfinfo/info.py#L449
42 Parameters
43 ----------
44 path:
45 Absolute path to the PDF file.
46 log:
47 Logger for warnings. Falls back to the module-level logger when omitted.
49 Returns
50 -------
51 bool
52 ``True`` when the PDF is tagged, ``False`` otherwise or on any error.
53 """
54 import pikepdf
56 _log = log or logger
57 try:
58 with pikepdf.open(path) as pdf:
59 mark_info = pdf.Root.get("/MarkInfo")
60 if mark_info is None:
61 return False
62 return bool(mark_info.get("/Marked", False))
63 except Exception:
64 _log.warning("Could not check PDF tag status for %s", path, exc_info=True)
65 return False
68def extract_pdf_text(
69 path: Path,
70 log: logging.Logger | None = None,
71) -> str | None:
72 """Run pdftotext on *path* and return the extracted text, or None on failure.
74 Parameters
75 ----------
76 path:
77 Absolute path to the PDF file.
78 log:
79 Logger for warnings. Falls back to the module-level logger when omitted.
81 Returns
82 -------
83 str | None
84 Extracted text, or ``None`` if pdftotext fails or the file is not a PDF.
85 """
86 from documents.utils import run_subprocess
88 _log = log or logger
89 try:
90 with tempfile.TemporaryDirectory() as tmpdir:
91 out_path = Path(tmpdir) / "text.txt"
92 run_subprocess(
93 [
94 "pdftotext",
95 "-q",
96 "-layout",
97 "-enc",
98 "UTF-8",
99 str(path),
100 str(out_path),
101 ],
102 logger=_log,
103 )
104 text = read_file_handle_unicode_errors(out_path, log=_log)
105 return text or None
106 except Exception:
107 _log.warning(
108 "Error while getting text from PDF document with pdftotext",
109 exc_info=True,
110 )
111 return None
114def post_process_text(text: str | None) -> str | None:
115 """Normalize extracted PDF/OCR text: collapse whitespace, strip padding.
117 Returns ``None`` for ``None`` or whitespace-only input, so callers can
118 treat "no text" and "only layout padding" the same way.
119 """
120 if not text:
121 return None
123 collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
124 no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
125 no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
127 # replace \0 prevents issues with saving to postgres.
128 # text may contain \0 when this character is present in PDF files.
129 result = no_trailing_whitespace.strip().replace("\0", " ")
130 return result or None
133def is_born_digital_text(
134 text: str | None,
135 path: Path,
136 log: logging.Logger | None = None,
137) -> bool:
138 """Decide whether already-extracted, normalized PDF text counts as born-digital.
140 This is the single source of truth for "does this PDF already have real
141 text", used both to decide whether to produce an archive file and to
142 decide whether OCR can be skipped. Both decisions must agree, or a
143 tagged-but-textless PDF can end up with no archive AND a forced OCR pass
144 (see GH #13387): raw ``pdftotext -layout`` output can be non-empty
145 (whitespace/form-feed padding) even when there is no real content, so
146 *text* must already be normalized via :func:`post_process_text`, not the
147 raw extraction.
149 Parameters
150 ----------
151 text:
152 The normalized extracted text (or ``None``) to evaluate.
153 path:
154 Absolute path to the PDF file, used for the tagged-PDF check.
155 log:
156 Logger for warnings. Falls back to the module-level logger when omitted.
158 Returns
159 -------
160 bool
161 Whether the PDF counts as born-digital (has real text, and is either
162 tagged or exceeds ``PDF_TEXT_MIN_LENGTH``).
163 """
164 if not text:
165 return False
166 return is_tagged_pdf(path, log=log) or len(text) > PDF_TEXT_MIN_LENGTH
169def pdf_born_digital_text(
170 path: Path,
171 log: logging.Logger | None = None,
172) -> tuple[str | None, bool]:
173 """Extract a PDF's text and decide whether it should be treated as born-digital.
175 Convenience wrapper around :func:`is_born_digital_text` for callers that
176 don't already have the PDF's text extracted (e.g. the archive-generation
177 decision, which runs before any parser has touched the file).
179 Parameters
180 ----------
181 path:
182 Absolute path to the PDF file.
183 log:
184 Logger for warnings. Falls back to the module-level logger when omitted.
186 Returns
187 -------
188 tuple[str | None, bool]
189 The normalized extracted text (or ``None``), and whether the PDF
190 counts as born-digital.
191 """
192 text = post_process_text(extract_pdf_text(path, log=log))
193 return text, is_born_digital_text(text, path, log=log)
196def read_file_handle_unicode_errors(
197 filepath: Path,
198 log: logging.Logger | None = None,
199) -> str:
200 """Read a file as text, detecting encoding via BOM and stripping NUL bytes.
202 Parameters
203 ----------
204 filepath:
205 Absolute path to the file to read.
206 log:
207 Logger to use for warnings. Falls back to the module-level logger
208 when omitted.
210 Returns
211 -------
212 str
213 File content as a string, with NUL bytes removed so the result is
214 safe to store in PostgreSQL text fields.
215 """
216 _log = log or logger
217 raw = filepath.read_bytes()
219 if raw.startswith((codecs.BOM_UTF16_LE, codecs.BOM_UTF16_BE)):
220 encoding = "utf-16"
221 elif raw.startswith(codecs.BOM_UTF8):
222 encoding = "utf-8-sig"
223 else:
224 encoding = "utf-8"
226 try:
227 text = raw.decode(encoding)
228 except UnicodeDecodeError as e:
229 _log.warning("Unicode error during text reading, continuing: %s", e)
230 text = raw.decode("utf-8", errors="replace")
232 # PostgreSQL rejects NUL (0x00) bytes in text fields
233 return text.replace("\x00", "")
236def get_page_count_for_pdf(
237 document_path: Path,
238 log: logging.Logger | None = None,
239) -> int | None:
240 """Return the number of pages in a PDF file using pikepdf.
242 Parameters
243 ----------
244 document_path:
245 Absolute path to the PDF file.
246 log:
247 Logger to use for warnings. Falls back to the module-level logger
248 when omitted.
250 Returns
251 -------
252 int | None
253 Page count, or ``None`` if the file cannot be opened or is not a
254 valid PDF.
255 """
256 import pikepdf
258 _log = log or logger
260 try:
261 with pikepdf.Pdf.open(document_path) as pdf:
262 return len(pdf.pages)
263 except Exception as e:
264 _log.warning("Unable to determine PDF page count for %s: %s", document_path, e)
265 return None
268def extract_pdf_metadata(
269 document_path: Path,
270 log: logging.Logger | None = None,
271) -> list[MetadataEntry]:
272 """Extract XMP/PDF metadata from a PDF file using pikepdf.
274 Reads all XMP metadata entries from the document and returns them as a
275 list of ``MetadataEntry`` dicts. The method never raises — any failure
276 to open the file or read a specific key is logged and skipped.
278 Parameters
279 ----------
280 document_path:
281 Absolute path to the PDF file.
282 log:
283 Logger to use for warnings and debug messages. Falls back to the
284 module-level logger when omitted.
286 Returns
287 -------
288 list[MetadataEntry]
289 Zero or more metadata entries. Returns ``[]`` if the file cannot
290 be opened or contains no readable XMP metadata.
291 """
292 import pikepdf
294 from paperless.parsers import MetadataEntry
296 _log = log or logger
297 result: list[MetadataEntry] = []
298 namespace_pattern = re.compile(r"\{(.*)\}(.*)")
300 try:
301 pdf = pikepdf.open(document_path)
302 meta = pdf.open_metadata()
303 except Exception as e:
304 _log.warning("Could not open PDF metadata for %s: %s", document_path, e)
305 return []
307 for key, value in meta.items():
308 if isinstance(value, list):
309 value = " ".join(str(e) for e in value)
310 value = str(value)
312 try:
313 m = namespace_pattern.match(key)
314 if m is None:
315 continue
317 namespace = m.group(1)
318 key_value = m.group(2)
320 try:
321 namespace.encode("utf-8")
322 key_value.encode("utf-8")
323 except UnicodeEncodeError as enc_err: # pragma: no cover
324 _log.debug("Skipping metadata key %s: %s", key, enc_err)
325 continue
327 result.append(
328 MetadataEntry(
329 namespace=namespace,
330 prefix=meta.REVERSE_NS[namespace],
331 key=key_value,
332 value=value,
333 ),
334 )
335 except Exception as e:
336 _log.warning(
337 "Error reading metadata key %s value %s: %s",
338 key,
339 value,
340 e,
341 )
343 return result