Coverage for paperless/parsers/tika.py: 31%

130 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1""" 

2Built-in Tika document parser. 

3 

4Handles Office documents (DOCX, ODT, XLS, XLSX, PPT, PPTX, RTF, etc.) by 

5sending them to an Apache Tika server for text extraction and a Gotenberg 

6server for PDF conversion. Because the source formats cannot be rendered by 

7a browser natively, the parser always produces a PDF rendition for display. 

8""" 

9 

10from __future__ import annotations 

11 

12import logging 

13import shutil 

14import tempfile 

15from contextlib import ExitStack 

16from pathlib import Path 

17from typing import TYPE_CHECKING 

18from typing import Self 

19 

20import httpx 

21from django.conf import settings 

22from django.utils import timezone 

23from gotenberg_client import GotenbergClient 

24from gotenberg_client.options import PdfAFormat 

25from tika_client import TikaClient 

26 

27from documents.parsers import ParseError 

28from documents.parsers import make_thumbnail_from_pdf 

29from paperless.config import OutputTypeConfig 

30from paperless.models import OutputTypeChoices 

31from paperless.version import __full_version_str__ 

32 

33if TYPE_CHECKING: 33 ↛ 34line 33 didn't jump to line 34 because the condition on line 33 was never true

34 import datetime 

35 from types import TracebackType 

36 

37 from paperless.parsers import MetadataEntry 

38 from paperless.parsers import ParserContext 

39 

40logger = logging.getLogger("paperless.parsing.tika") 

41 

42_SUPPORTED_MIME_TYPES: dict[str, str] = { 

43 "application/msword": ".doc", 

44 "application/vnd.openxmlformats-officedocument.wordprocessingml.document": ".docx", 

45 "application/vnd.ms-excel": ".xls", 

46 "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": ".xlsx", 

47 "application/vnd.ms-powerpoint": ".ppt", 

48 "application/vnd.openxmlformats-officedocument.presentationml.presentation": ".pptx", 

49 "application/vnd.openxmlformats-officedocument.presentationml.slideshow": ".ppsx", 

50 "application/vnd.oasis.opendocument.presentation": ".odp", 

51 "application/vnd.oasis.opendocument.spreadsheet": ".ods", 

52 "application/vnd.oasis.opendocument.text": ".odt", 

53 "application/vnd.oasis.opendocument.graphics": ".odg", 

54 "text/rtf": ".rtf", 

55} 

56 

57 

58class TikaDocumentParser: 

59 """Parse Office documents via Apache Tika and Gotenberg for Paperless-ngx. 

60 

61 Text extraction is handled by the Tika server. PDF conversion for display 

62 is handled by Gotenberg (LibreOffice route). Because the source formats 

63 cannot be rendered by a browser natively, ``requires_pdf_rendition`` is 

64 True and the PDF is always produced regardless of the ``produce_archive`` 

65 flag passed to ``parse``. 

66 

67 Both ``TikaClient`` and ``GotenbergClient`` are opened once in 

68 ``__enter__`` via an ``ExitStack`` and shared across ``parse``, 

69 ``extract_metadata``, and ``_convert_to_pdf`` calls, then closed via 

70 ``ExitStack.close()`` in ``__exit__``. The parser must always be used 

71 as a context manager. 

72 

73 Class attributes 

74 ---------------- 

75 name : str 

76 Human-readable parser name. 

77 version : str 

78 Semantic version string, kept in sync with Paperless-ngx releases. 

79 author : str 

80 Maintainer name. 

81 url : str 

82 Issue tracker / source URL. 

83 """ 

84 

85 name: str = "Paperless-ngx Tika Parser" 

86 version: str = __full_version_str__ 

87 author: str = "Paperless-ngx Contributors" 

88 url: str = "https://github.com/paperless-ngx/paperless-ngx" 

89 

90 # ------------------------------------------------------------------ 

91 # Class methods 

92 # ------------------------------------------------------------------ 

93 

94 @classmethod 

95 def supported_mime_types(cls) -> dict[str, str]: 

96 """Return the MIME types this parser handles. 

97 

98 Returns 

99 ------- 

100 dict[str, str] 

101 Mapping of MIME type to preferred file extension. 

102 """ 

103 return _SUPPORTED_MIME_TYPES 

104 

105 @classmethod 

106 def score( 

107 cls, 

108 mime_type: str, 

109 filename: str, 

110 path: Path | None = None, 

111 ) -> int | None: 

112 """Return the priority score for handling this file. 

113 

114 Returns ``None`` when Tika integration is disabled so the registry 

115 skips this parser entirely. 

116 

117 Parameters 

118 ---------- 

119 mime_type: 

120 Detected MIME type of the file. 

121 filename: 

122 Original filename including extension. 

123 path: 

124 Optional filesystem path. Not inspected by this parser. 

125 

126 Returns 

127 ------- 

128 int | None 

129 10 if TIKA_ENABLED and the MIME type is supported, otherwise None. 

130 """ 

131 if not settings.TIKA_ENABLED: 

132 return None 

133 if mime_type in _SUPPORTED_MIME_TYPES: 

134 return 10 

135 return None 

136 

137 # ------------------------------------------------------------------ 

138 # Properties 

139 # ------------------------------------------------------------------ 

140 

141 @property 

142 def can_produce_archive(self) -> bool: 

143 """Whether this parser can produce a searchable PDF archive copy. 

144 

145 Returns 

146 ------- 

147 bool 

148 Always False — Tika produces a display PDF, not an OCR archive. 

149 """ 

150 return False 

151 

152 @property 

153 def requires_pdf_rendition(self) -> bool: 

154 """Whether the parser must produce a PDF for the frontend to display. 

155 

156 Returns 

157 ------- 

158 bool 

159 Always True — Office formats cannot be rendered natively in a 

160 browser, so a PDF conversion is always required for display. 

161 """ 

162 return True 

163 

164 # ------------------------------------------------------------------ 

165 # Lifecycle 

166 # ------------------------------------------------------------------ 

167 

168 def __init__(self, logging_group: object = None) -> None: 

169 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True) 

170 self._tempdir = Path( 

171 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR), 

172 ) 

173 self._text: str | None = None 

174 self._date: datetime.datetime | None = None 

175 self._archive_path: Path | None = None 

176 self._exit_stack = ExitStack() 

177 self._tika_client: TikaClient | None = None 

178 self._gotenberg_client: GotenbergClient | None = None 

179 

180 def __enter__(self) -> Self: 

181 self._tika_client = self._exit_stack.enter_context( 

182 TikaClient( 

183 tika_url=settings.TIKA_ENDPOINT, 

184 timeout=settings.CELERY_TASK_TIME_LIMIT, 

185 ), 

186 ) 

187 self._gotenberg_client = self._exit_stack.enter_context( 

188 GotenbergClient( 

189 host=settings.TIKA_GOTENBERG_ENDPOINT, 

190 timeout=settings.CELERY_TASK_TIME_LIMIT, 

191 ), 

192 ) 

193 return self 

194 

195 def __exit__( 

196 self, 

197 exc_type: type[BaseException] | None, 

198 exc_val: BaseException | None, 

199 exc_tb: TracebackType | None, 

200 ) -> None: 

201 self._exit_stack.close() 

202 logger.debug("Cleaning up temporary directory %s", self._tempdir) 

203 shutil.rmtree(self._tempdir, ignore_errors=True) 

204 

205 # ------------------------------------------------------------------ 

206 # Core parsing interface 

207 # ------------------------------------------------------------------ 

208 

209 def configure(self, context: ParserContext) -> None: 

210 pass 

211 

212 def parse( 

213 self, 

214 document_path: Path, 

215 mime_type: str, 

216 *, 

217 produce_archive: bool = True, 

218 ) -> None: 

219 """Send the document to Tika for text extraction and Gotenberg for PDF. 

220 

221 Because ``requires_pdf_rendition`` is True the PDF conversion is 

222 always performed — the ``produce_archive`` flag is intentionally 

223 ignored. 

224 

225 Parameters 

226 ---------- 

227 document_path: 

228 Absolute path to the document file to parse. 

229 mime_type: 

230 Detected MIME type of the document. 

231 produce_archive: 

232 Accepted for protocol compatibility but ignored; the PDF rendition 

233 is always produced since the source format cannot be displayed 

234 natively in the browser. 

235 

236 Raises 

237 ------ 

238 documents.parsers.ParseError 

239 If Tika or Gotenberg returns an error. 

240 """ 

241 if TYPE_CHECKING: 

242 assert self._tika_client is not None 

243 

244 logger.info("Sending %s to Tika server", document_path) 

245 

246 try: 

247 try: 

248 parsed = self._tika_client.tika.as_text.from_file( 

249 document_path, 

250 mime_type, 

251 ) 

252 except httpx.HTTPStatusError as err: 

253 # Workaround https://issues.apache.org/jira/browse/TIKA-4110 

254 # Tika fails with some files as multi-part form data 

255 if err.response.status_code == httpx.codes.INTERNAL_SERVER_ERROR: 

256 parsed = self._tika_client.tika.as_text.from_buffer( 

257 document_path.read_bytes(), 

258 mime_type, 

259 ) 

260 else: # pragma: no cover 

261 raise 

262 except Exception as err: 

263 raise ParseError( 

264 f"Could not parse {document_path} with tika server at " 

265 f"{settings.TIKA_ENDPOINT}: {err}", 

266 ) from err 

267 

268 self._text = (parsed.content or "").strip() 

269 

270 self._date = parsed.created 

271 if self._date is not None and timezone.is_naive(self._date): 

272 self._date = timezone.make_aware(self._date) 

273 

274 # Always convert — requires_pdf_rendition=True means the browser 

275 # cannot display the source format natively. 

276 self._archive_path = self._convert_to_pdf(document_path) 

277 

278 # ------------------------------------------------------------------ 

279 # Result accessors 

280 # ------------------------------------------------------------------ 

281 

282 def get_text(self) -> str: 

283 """Return the plain-text content extracted during parse. 

284 

285 Returns 

286 ------- 

287 str 

288 Extracted text, or an empty string if no text could be found. 

289 """ 

290 return self._text or "" 

291 

292 def get_date(self) -> datetime.datetime | None: 

293 """Return the document date detected during parse. 

294 

295 Returns 

296 ------- 

297 datetime.datetime | None 

298 Creation date from Tika metadata, or None if not detected. 

299 """ 

300 return self._date 

301 

302 def get_archive_path(self) -> Path | None: 

303 """Return the path to the generated PDF rendition, or None. 

304 

305 Returns 

306 ------- 

307 Path | None 

308 Path to the PDF produced by Gotenberg, or None if parse has not 

309 been called yet. 

310 """ 

311 return self._archive_path 

312 

313 # ------------------------------------------------------------------ 

314 # Thumbnail and metadata 

315 # ------------------------------------------------------------------ 

316 

317 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path: 

318 """Generate a thumbnail from the PDF rendition of the document. 

319 

320 Converts the document to PDF first if not already done. 

321 

322 Parameters 

323 ---------- 

324 document_path: 

325 Absolute path to the source document. 

326 mime_type: 

327 Detected MIME type of the document. 

328 

329 Returns 

330 ------- 

331 Path 

332 Path to the generated WebP thumbnail inside the temporary directory. 

333 """ 

334 if self._archive_path is None: 

335 self._archive_path = self._convert_to_pdf(document_path) 

336 return make_thumbnail_from_pdf(self._archive_path, self._tempdir) 

337 

338 def get_page_count( 

339 self, 

340 document_path: Path, 

341 mime_type: str, 

342 ) -> int | None: 

343 """Return the number of pages in the document. 

344 

345 Counts pages in the archive PDF produced by a preceding parse() 

346 call. Returns ``None`` if parse() has not been called yet or if 

347 no archive was produced. 

348 

349 Returns 

350 ------- 

351 int | None 

352 Page count of the archive PDF, or ``None``. 

353 """ 

354 if self._archive_path is not None: 

355 from paperless.parsers.utils import get_page_count_for_pdf 

356 

357 return get_page_count_for_pdf(self._archive_path, log=logger) 

358 return None 

359 

360 def extract_metadata( 

361 self, 

362 document_path: Path, 

363 mime_type: str, 

364 ) -> list[MetadataEntry]: 

365 """Extract format-specific metadata via the Tika metadata endpoint. 

366 

367 Returns 

368 ------- 

369 list[MetadataEntry] 

370 All key/value pairs returned by Tika, or ``[]`` on error. 

371 """ 

372 if TYPE_CHECKING: 

373 assert self._tika_client is not None 

374 

375 try: 

376 parsed = self._tika_client.metadata.from_file(document_path, mime_type) 

377 return [ 

378 { 

379 "namespace": "", 

380 "prefix": "", 

381 "key": key, 

382 "value": parsed.data[key], 

383 } 

384 for key in parsed.data 

385 ] 

386 except Exception as e: 

387 logger.warning( 

388 "Error while fetching document metadata for %s: %s", 

389 document_path, 

390 e, 

391 ) 

392 return [] 

393 

394 # ------------------------------------------------------------------ 

395 # Private helpers 

396 # ------------------------------------------------------------------ 

397 

398 def _convert_to_pdf(self, document_path: Path) -> Path: 

399 """Convert the document to PDF using Gotenberg's LibreOffice route. 

400 

401 Parameters 

402 ---------- 

403 document_path: 

404 Absolute path to the source document. 

405 

406 Returns 

407 ------- 

408 Path 

409 Path to the generated PDF inside the temporary directory. 

410 

411 Raises 

412 ------ 

413 documents.parsers.ParseError 

414 If Gotenberg returns an error. 

415 """ 

416 if TYPE_CHECKING: 

417 assert self._gotenberg_client is not None 

418 

419 pdf_path = self._tempdir / "convert.pdf" 

420 

421 logger.info("Converting %s to PDF as %s", document_path, pdf_path) 

422 

423 with self._gotenberg_client.libre_office.to_pdf() as route: 

424 # Preserve document fields as authored. updateIndexes (Gotenberg's 

425 # default) triggers a refresh() that rewrites dynamic fields like 

426 # auto-dates to the current date. 

427 route.update_indexes(update_indexes=False) 

428 

429 # Set the output format of the resulting PDF. 

430 # OutputTypeConfig reads the database-stored ApplicationConfiguration 

431 # first, then falls back to the PAPERLESS_OCR_OUTPUT_TYPE env var. 

432 output_type = OutputTypeConfig().output_type 

433 if output_type in { 

434 OutputTypeChoices.PDF_A, 

435 OutputTypeChoices.PDF_A2, 

436 }: 

437 route.pdf_format(PdfAFormat.A2b) 

438 elif output_type == OutputTypeChoices.PDF_A1: 

439 logger.warning( 

440 "Gotenberg does not support PDF/A-1a, choosing PDF/A-2b instead", 

441 ) 

442 route.pdf_format(PdfAFormat.A2b) 

443 elif output_type == OutputTypeChoices.PDF_A3: 

444 route.pdf_format(PdfAFormat.A3b) 

445 

446 route.convert(document_path) 

447 

448 try: 

449 response = route.run() 

450 pdf_path.write_bytes(response.content) 

451 return pdf_path 

452 except Exception as err: 

453 raise ParseError( 

454 f"Error while converting document to PDF: {err}", 

455 ) from err