Coverage for paperless/parsers/utils.py: 16%

106 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1""" 

2Shared utilities for Paperless-ngx document parsers. 

3 

4Functions here are format-neutral helpers that multiple parsers need. 

5Keeping them here avoids parsers inheriting from each other just to 

6share implementation. 

7""" 

8 

9from __future__ import annotations 

10 

11import codecs 

12import logging 

13import re 

14import tempfile 

15from pathlib import Path 

16from typing import TYPE_CHECKING 

17from typing import Final 

18 

19if TYPE_CHECKING: 19 ↛ 20line 19 didn't jump to line 20 because the condition on line 19 was never true

20 from paperless.parsers import MetadataEntry 

21 

22logger = logging.getLogger("paperless.parsers.utils") 

23 

24# Minimum character count for a PDF to be considered "born-digital" (has real text). 

25# Used by both the consumer (archive decision) and the tesseract parser (skip-OCR decision). 

26PDF_TEXT_MIN_LENGTH: Final[int] = 50 

27 

28 

29def is_tagged_pdf( 

30 path: Path, 

31 log: logging.Logger | None = None, 

32) -> bool: 

33 """Return True if the PDF declares itself as tagged (born-digital indicator). 

34 

35 Tagged PDFs (e.g. exported from Word or LibreOffice) have ``/MarkInfo`` 

36 with ``/Marked true`` in the document root. This is a reliable signal 

37 that the document has a logical structure and embedded text — running OCR 

38 on it is unnecessary and archive generation can be skipped. 

39 

40 https://github.com/ocrmypdf/OCRmyPDF/blob/4e974ebd465a5921b2e79004f098f5d203010282/src/ocrmypdf/pdfinfo/info.py#L449 

41 

42 Parameters 

43 ---------- 

44 path: 

45 Absolute path to the PDF file. 

46 log: 

47 Logger for warnings. Falls back to the module-level logger when omitted. 

48 

49 Returns 

50 ------- 

51 bool 

52 ``True`` when the PDF is tagged, ``False`` otherwise or on any error. 

53 """ 

54 import pikepdf 

55 

56 _log = log or logger 

57 try: 

58 with pikepdf.open(path) as pdf: 

59 mark_info = pdf.Root.get("/MarkInfo") 

60 if mark_info is None: 

61 return False 

62 return bool(mark_info.get("/Marked", False)) 

63 except Exception: 

64 _log.warning("Could not check PDF tag status for %s", path, exc_info=True) 

65 return False 

66 

67 

68def extract_pdf_text( 

69 path: Path, 

70 log: logging.Logger | None = None, 

71) -> str | None: 

72 """Run pdftotext on *path* and return the extracted text, or None on failure. 

73 

74 Parameters 

75 ---------- 

76 path: 

77 Absolute path to the PDF file. 

78 log: 

79 Logger for warnings. Falls back to the module-level logger when omitted. 

80 

81 Returns 

82 ------- 

83 str | None 

84 Extracted text, or ``None`` if pdftotext fails or the file is not a PDF. 

85 """ 

86 from documents.utils import run_subprocess 

87 

88 _log = log or logger 

89 try: 

90 with tempfile.TemporaryDirectory() as tmpdir: 

91 out_path = Path(tmpdir) / "text.txt" 

92 run_subprocess( 

93 [ 

94 "pdftotext", 

95 "-q", 

96 "-layout", 

97 "-enc", 

98 "UTF-8", 

99 str(path), 

100 str(out_path), 

101 ], 

102 logger=_log, 

103 ) 

104 text = read_file_handle_unicode_errors(out_path, log=_log) 

105 return text or None 

106 except Exception: 

107 _log.warning( 

108 "Error while getting text from PDF document with pdftotext", 

109 exc_info=True, 

110 ) 

111 return None 

112 

113 

114def post_process_text(text: str | None) -> str | None: 

115 """Normalize extracted PDF/OCR text: collapse whitespace, strip padding. 

116 

117 Returns ``None`` for ``None`` or whitespace-only input, so callers can 

118 treat "no text" and "only layout padding" the same way. 

119 """ 

120 if not text: 

121 return None 

122 

123 collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text) 

124 no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces) 

125 no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace) 

126 

127 # replace \0 prevents issues with saving to postgres. 

128 # text may contain \0 when this character is present in PDF files. 

129 result = no_trailing_whitespace.strip().replace("\0", " ") 

130 return result or None 

131 

132 

133def is_born_digital_text( 

134 text: str | None, 

135 path: Path, 

136 log: logging.Logger | None = None, 

137) -> bool: 

138 """Decide whether already-extracted, normalized PDF text counts as born-digital. 

139 

140 This is the single source of truth for "does this PDF already have real 

141 text", used both to decide whether to produce an archive file and to 

142 decide whether OCR can be skipped. Both decisions must agree, or a 

143 tagged-but-textless PDF can end up with no archive AND a forced OCR pass 

144 (see GH #13387): raw ``pdftotext -layout`` output can be non-empty 

145 (whitespace/form-feed padding) even when there is no real content, so 

146 *text* must already be normalized via :func:`post_process_text`, not the 

147 raw extraction. 

148 

149 Parameters 

150 ---------- 

151 text: 

152 The normalized extracted text (or ``None``) to evaluate. 

153 path: 

154 Absolute path to the PDF file, used for the tagged-PDF check. 

155 log: 

156 Logger for warnings. Falls back to the module-level logger when omitted. 

157 

158 Returns 

159 ------- 

160 bool 

161 Whether the PDF counts as born-digital (has real text, and is either 

162 tagged or exceeds ``PDF_TEXT_MIN_LENGTH``). 

163 """ 

164 if not text: 

165 return False 

166 return is_tagged_pdf(path, log=log) or len(text) > PDF_TEXT_MIN_LENGTH 

167 

168 

169def pdf_born_digital_text( 

170 path: Path, 

171 log: logging.Logger | None = None, 

172) -> tuple[str | None, bool]: 

173 """Extract a PDF's text and decide whether it should be treated as born-digital. 

174 

175 Convenience wrapper around :func:`is_born_digital_text` for callers that 

176 don't already have the PDF's text extracted (e.g. the archive-generation 

177 decision, which runs before any parser has touched the file). 

178 

179 Parameters 

180 ---------- 

181 path: 

182 Absolute path to the PDF file. 

183 log: 

184 Logger for warnings. Falls back to the module-level logger when omitted. 

185 

186 Returns 

187 ------- 

188 tuple[str | None, bool] 

189 The normalized extracted text (or ``None``), and whether the PDF 

190 counts as born-digital. 

191 """ 

192 text = post_process_text(extract_pdf_text(path, log=log)) 

193 return text, is_born_digital_text(text, path, log=log) 

194 

195 

196def read_file_handle_unicode_errors( 

197 filepath: Path, 

198 log: logging.Logger | None = None, 

199) -> str: 

200 """Read a file as text, detecting encoding via BOM and stripping NUL bytes. 

201 

202 Parameters 

203 ---------- 

204 filepath: 

205 Absolute path to the file to read. 

206 log: 

207 Logger to use for warnings. Falls back to the module-level logger 

208 when omitted. 

209 

210 Returns 

211 ------- 

212 str 

213 File content as a string, with NUL bytes removed so the result is 

214 safe to store in PostgreSQL text fields. 

215 """ 

216 _log = log or logger 

217 raw = filepath.read_bytes() 

218 

219 if raw.startswith((codecs.BOM_UTF16_LE, codecs.BOM_UTF16_BE)): 

220 encoding = "utf-16" 

221 elif raw.startswith(codecs.BOM_UTF8): 

222 encoding = "utf-8-sig" 

223 else: 

224 encoding = "utf-8" 

225 

226 try: 

227 text = raw.decode(encoding) 

228 except UnicodeDecodeError as e: 

229 _log.warning("Unicode error during text reading, continuing: %s", e) 

230 text = raw.decode("utf-8", errors="replace") 

231 

232 # PostgreSQL rejects NUL (0x00) bytes in text fields 

233 return text.replace("\x00", "") 

234 

235 

236def get_page_count_for_pdf( 

237 document_path: Path, 

238 log: logging.Logger | None = None, 

239) -> int | None: 

240 """Return the number of pages in a PDF file using pikepdf. 

241 

242 Parameters 

243 ---------- 

244 document_path: 

245 Absolute path to the PDF file. 

246 log: 

247 Logger to use for warnings. Falls back to the module-level logger 

248 when omitted. 

249 

250 Returns 

251 ------- 

252 int | None 

253 Page count, or ``None`` if the file cannot be opened or is not a 

254 valid PDF. 

255 """ 

256 import pikepdf 

257 

258 _log = log or logger 

259 

260 try: 

261 with pikepdf.Pdf.open(document_path) as pdf: 

262 return len(pdf.pages) 

263 except Exception as e: 

264 _log.warning("Unable to determine PDF page count for %s: %s", document_path, e) 

265 return None 

266 

267 

268def extract_pdf_metadata( 

269 document_path: Path, 

270 log: logging.Logger | None = None, 

271) -> list[MetadataEntry]: 

272 """Extract XMP/PDF metadata from a PDF file using pikepdf. 

273 

274 Reads all XMP metadata entries from the document and returns them as a 

275 list of ``MetadataEntry`` dicts. The method never raises — any failure 

276 to open the file or read a specific key is logged and skipped. 

277 

278 Parameters 

279 ---------- 

280 document_path: 

281 Absolute path to the PDF file. 

282 log: 

283 Logger to use for warnings and debug messages. Falls back to the 

284 module-level logger when omitted. 

285 

286 Returns 

287 ------- 

288 list[MetadataEntry] 

289 Zero or more metadata entries. Returns ``[]`` if the file cannot 

290 be opened or contains no readable XMP metadata. 

291 """ 

292 import pikepdf 

293 

294 from paperless.parsers import MetadataEntry 

295 

296 _log = log or logger 

297 result: list[MetadataEntry] = [] 

298 namespace_pattern = re.compile(r"\{(.*)\}(.*)") 

299 

300 try: 

301 pdf = pikepdf.open(document_path) 

302 meta = pdf.open_metadata() 

303 except Exception as e: 

304 _log.warning("Could not open PDF metadata for %s: %s", document_path, e) 

305 return [] 

306 

307 for key, value in meta.items(): 

308 if isinstance(value, list): 

309 value = " ".join(str(e) for e in value) 

310 value = str(value) 

311 

312 try: 

313 m = namespace_pattern.match(key) 

314 if m is None: 

315 continue 

316 

317 namespace = m.group(1) 

318 key_value = m.group(2) 

319 

320 try: 

321 namespace.encode("utf-8") 

322 key_value.encode("utf-8") 

323 except UnicodeEncodeError as enc_err: # pragma: no cover 

324 _log.debug("Skipping metadata key %s: %s", key, enc_err) 

325 continue 

326 

327 result.append( 

328 MetadataEntry( 

329 namespace=namespace, 

330 prefix=meta.REVERSE_NS[namespace], 

331 key=key_value, 

332 value=value, 

333 ), 

334 ) 

335 except Exception as e: 

336 _log.warning( 

337 "Error reading metadata key %s value %s: %s", 

338 key, 

339 value, 

340 e, 

341 ) 

342 

343 return result