Coverage for paperless/parsers/remote.py: 39%

133 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1""" 

2Built-in remote-OCR document parser. 

3 

4Handles documents by sending them to a configured remote OCR engine 

5(currently Azure AI Vision / Document Intelligence) and retrieving both 

6the extracted text and a searchable PDF with an embedded text layer. For 

7born-digital PDFs that need no archive copy, the remote call is skipped 

8entirely in favor of locally-extracted text (see ``RemoteDocumentParser.parse``). 

9 

10When no engine is configured, ``score()`` returns ``None`` so the parser 

11is effectively invisible to the registry — the tesseract parser handles 

12these MIME types instead. 

13""" 

14 

15from __future__ import annotations 

16 

17import logging 

18import shutil 

19import tempfile 

20from pathlib import Path 

21from typing import TYPE_CHECKING 

22from typing import Self 

23 

24from django.conf import settings 

25 

26from documents.parsers import ParseError 

27from paperless.parsers.utils import extract_pdf_text 

28from paperless.parsers.utils import post_process_text 

29from paperless.version import __full_version_str__ 

30 

31if TYPE_CHECKING: 31 ↛ 32line 31 didn't jump to line 32 because the condition on line 31 was never true

32 import datetime 

33 from types import TracebackType 

34 

35 from azure.core.pipeline import PipelineRequest 

36 

37 from paperless.parsers import MetadataEntry 

38 from paperless.parsers import ParserContext 

39 

40logger = logging.getLogger("paperless.parsing.remote") 

41 

42_SUPPORTED_MIME_TYPES: dict[str, str] = { 

43 "application/pdf": ".pdf", 

44 "image/png": ".png", 

45 "image/jpeg": ".jpg", 

46 "image/tiff": ".tiff", 

47 "image/bmp": ".bmp", 

48 "image/gif": ".gif", 

49 "image/webp": ".webp", 

50} 

51 

52 

53class RemoteEngineConfig: 

54 """Holds and validates the remote OCR engine configuration.""" 

55 

56 def __init__( 

57 self, 

58 engine: str | None, 

59 api_key: str | None = None, 

60 endpoint: str | None = None, 

61 ) -> None: 

62 self.engine = engine 

63 self.api_key = api_key 

64 self.endpoint = endpoint 

65 

66 @classmethod 

67 def from_app_config(cls) -> Self: 

68 """Build the config from the app config, falling back to the env.""" 

69 from paperless.config import RemoteOCRConfig 

70 

71 app_config = RemoteOCRConfig() 

72 return cls( 

73 engine=app_config.remote_ocr_engine, 

74 api_key=app_config.remote_ocr_api_key, 

75 endpoint=app_config.remote_ocr_endpoint, 

76 ) 

77 

78 def engine_is_valid(self) -> bool: 

79 """Return True when the engine is known and fully configured.""" 

80 return ( 

81 self.engine in ("azureai",) 

82 and self.api_key is not None 

83 and not (self.engine == "azureai" and self.endpoint is None) 

84 ) 

85 

86 

87class RemoteDocumentParser: 

88 """Parse documents via a remote OCR API (currently Azure AI Vision). 

89 

90 This parser sends documents to a remote engine that returns both 

91 extracted text and a searchable PDF with an embedded text layer, 

92 except when ``parse()`` is called with ``produce_archive=False`` for 

93 a PDF, in which case the remote call is skipped and only locally 

94 extracted text is returned (no archive). It does not depend on 

95 Tesseract or ocrmypdf. 

96 

97 Class attributes 

98 ---------------- 

99 name : str 

100 Human-readable parser name. 

101 version : str 

102 Semantic version string, kept in sync with Paperless-ngx releases. 

103 author : str 

104 Maintainer name. 

105 url : str 

106 Issue tracker / source URL. 

107 uses_remote_service : bool 

108 Content is sent to a remote service, True so that the registry 

109 can skip this parser if remote processing was not requested. 

110 """ 

111 

112 name: str = "Paperless-ngx Remote OCR Parser" 

113 version: str = __full_version_str__ 

114 author: str = "Paperless-ngx Contributors" 

115 url: str = "https://github.com/paperless-ngx/paperless-ngx" 

116 

117 uses_remote_service: bool = True 

118 

119 # ------------------------------------------------------------------ 

120 # Class methods 

121 # ------------------------------------------------------------------ 

122 

123 @classmethod 

124 def supported_mime_types(cls) -> dict[str, str]: 

125 """Return the MIME types this parser can handle. 

126 

127 The full set is always returned regardless of whether a remote 

128 engine is configured. The ``score()`` method handles the 

129 "am I active?" logic by returning ``None`` when not configured. 

130 

131 Returns 

132 ------- 

133 dict[str, str] 

134 Mapping of MIME type to preferred file extension. 

135 """ 

136 return _SUPPORTED_MIME_TYPES 

137 

138 @classmethod 

139 def score( 

140 cls, 

141 mime_type: str, 

142 filename: str, 

143 path: Path | None = None, 

144 ) -> int | None: 

145 """Return the priority score for handling this file, or None. 

146 

147 Returns ``None`` when no valid remote engine is configured, 

148 making the parser invisible to the registry for this file. 

149 When configured, returns 20 — higher than the Tesseract parser's 

150 default of 10 — so the remote engine takes priority. 

151 

152 Parameters 

153 ---------- 

154 mime_type: 

155 Detected MIME type of the file. 

156 filename: 

157 Original filename including extension. 

158 path: 

159 Optional filesystem path. Not inspected by this parser. 

160 

161 Returns 

162 ------- 

163 int | None 

164 20 when the remote engine is configured and the MIME type is 

165 supported, otherwise None. 

166 """ 

167 config = RemoteEngineConfig.from_app_config() 

168 if not config.engine_is_valid(): 168 ↛ 170line 168 didn't jump to line 170 because the condition on line 168 was always true

169 return None 

170 if mime_type not in _SUPPORTED_MIME_TYPES: 

171 return None 

172 return 20 

173 

174 # ------------------------------------------------------------------ 

175 # Properties 

176 # ------------------------------------------------------------------ 

177 

178 @property 

179 def can_produce_archive(self) -> bool: 

180 """Whether this parser can produce a searchable PDF archive copy. 

181 

182 Returns 

183 ------- 

184 bool 

185 Always True — the remote engine is capable of returning a PDF 

186 with an embedded text layer to serve as the archive copy. 

187 Whether it actually does so for a given document depends on 

188 ``produce_archive`` passed to :meth:`parse` (see there for when 

189 the remote engine call, and thus archive generation, is skipped). 

190 """ 

191 return True 

192 

193 @property 

194 def requires_pdf_rendition(self) -> bool: 

195 """Whether the parser must produce a PDF for the frontend to display. 

196 

197 Returns 

198 ------- 

199 bool 

200 Always False — all supported originals are displayable by 

201 the browser (PDF) or handled via the archive copy (images). 

202 """ 

203 return False 

204 

205 # ------------------------------------------------------------------ 

206 # Lifecycle 

207 # ------------------------------------------------------------------ 

208 

209 def __init__(self, logging_group: object = None) -> None: 

210 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True) 

211 self._tempdir = Path( 

212 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR), 

213 ) 

214 self._logging_group = logging_group 

215 self._text: str | None = None 

216 self._archive_path: Path | None = None 

217 

218 def __enter__(self) -> Self: 

219 return self 

220 

221 def __exit__( 

222 self, 

223 exc_type: type[BaseException] | None, 

224 exc_val: BaseException | None, 

225 exc_tb: TracebackType | None, 

226 ) -> None: 

227 logger.debug("Cleaning up temporary directory %s", self._tempdir) 

228 shutil.rmtree(self._tempdir, ignore_errors=True) 

229 

230 # ------------------------------------------------------------------ 

231 # Core parsing interface 

232 # ------------------------------------------------------------------ 

233 

234 def configure(self, context: ParserContext) -> None: 

235 pass 

236 

237 def parse( 

238 self, 

239 document_path: Path, 

240 mime_type: str, 

241 *, 

242 produce_archive: bool = True, 

243 ) -> None: 

244 """Send the document to the remote engine and store results. 

245 

246 When *produce_archive* is False for a PDF, the caller (via 

247 ``documents.consumer.should_produce_archive``) has already determined 

248 that the document is born-digital and needs no archive — skip the 

249 remote engine entirely rather than re-OCRing it and creating a 

250 duplicate text layer. 

251 

252 Parameters 

253 ---------- 

254 document_path: 

255 Absolute path to the document file to parse. 

256 mime_type: 

257 Detected MIME type of the document. 

258 produce_archive: 

259 Whether an archive copy is wanted. For PDFs, False skips the 

260 remote engine and uses locally-extracted text instead. 

261 """ 

262 config = RemoteEngineConfig.from_app_config() 

263 

264 if not config.engine_is_valid(): 

265 logger.warning( 

266 "No valid remote parser engine is configured, content will be empty.", 

267 ) 

268 self._text = "" 

269 return 

270 

271 if not produce_archive and mime_type == "application/pdf": 

272 logger.debug( 

273 "Remote OCR: skipped — no archive requested, " 

274 "using locally-extracted text", 

275 ) 

276 self._text = ( 

277 post_process_text(extract_pdf_text(document_path, log=logger)) or "" 

278 ) 

279 return 

280 

281 if config.engine == "azureai": 

282 self._text = self._azure_ai_vision_parse(document_path, config) 

283 

284 # ------------------------------------------------------------------ 

285 # Result accessors 

286 # ------------------------------------------------------------------ 

287 

288 def get_text(self) -> str: 

289 """Return the plain-text content extracted during parse.""" 

290 return self._text or "" 

291 

292 def get_date(self) -> datetime.datetime | None: 

293 """Return the document date detected during parse. 

294 

295 Returns 

296 ------- 

297 datetime.datetime | None 

298 Always None — the remote parser does not detect dates. 

299 """ 

300 return None 

301 

302 def get_archive_path(self) -> Path | None: 

303 """Return the path to the generated archive PDF, or None.""" 

304 return self._archive_path 

305 

306 # ------------------------------------------------------------------ 

307 # Thumbnail and metadata 

308 # ------------------------------------------------------------------ 

309 

310 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path: 

311 """Generate a thumbnail image for the document. 

312 

313 Uses the archive PDF produced by the remote engine when available, 

314 otherwise falls back to the original document path (PDF inputs). 

315 

316 Parameters 

317 ---------- 

318 document_path: 

319 Absolute path to the source document. 

320 mime_type: 

321 Detected MIME type of the document. 

322 

323 Returns 

324 ------- 

325 Path 

326 Path to the generated WebP thumbnail inside the temp directory. 

327 """ 

328 # make_thumbnail_from_pdf lives in documents.parsers for now; 

329 # it will move to paperless.parsers.utils when the tesseract 

330 # parser is migrated in a later phase. 

331 from documents.parsers import make_thumbnail_from_pdf 

332 

333 return make_thumbnail_from_pdf( 

334 self._archive_path or document_path, 

335 self._tempdir, 

336 self._logging_group, 

337 ) 

338 

339 def get_page_count( 

340 self, 

341 document_path: Path, 

342 mime_type: str, 

343 ) -> int | None: 

344 """Return the number of pages in a PDF document. 

345 

346 Parameters 

347 ---------- 

348 document_path: 

349 Absolute path to the source document. 

350 mime_type: 

351 Detected MIME type of the document. 

352 

353 Returns 

354 ------- 

355 int | None 

356 Page count for PDF inputs, or ``None`` for other MIME types. 

357 """ 

358 if mime_type != "application/pdf": 

359 return None 

360 

361 from paperless.parsers.utils import get_page_count_for_pdf 

362 

363 return get_page_count_for_pdf(document_path, log=logger) 

364 

365 def extract_metadata( 

366 self, 

367 document_path: Path, 

368 mime_type: str, 

369 ) -> list[MetadataEntry]: 

370 """Extract format-specific metadata from the document. 

371 

372 Delegates to the shared pikepdf-based extractor for PDF files. 

373 Returns ``[]`` for all other MIME types. 

374 

375 Parameters 

376 ---------- 

377 document_path: 

378 Absolute path to the file to extract metadata from. 

379 mime_type: 

380 MIME type of the file. May be ``"application/pdf"`` when 

381 called for the archive version of an image original. 

382 

383 Returns 

384 ------- 

385 list[MetadataEntry] 

386 Zero or more metadata entries. 

387 """ 

388 if mime_type != "application/pdf": 

389 return [] 

390 

391 from paperless.parsers.utils import extract_pdf_metadata 

392 

393 return extract_pdf_metadata(document_path, log=logger) 

394 

395 # ------------------------------------------------------------------ 

396 # Private helpers 

397 # ------------------------------------------------------------------ 

398 

399 def _azure_ai_vision_parse( 

400 self, 

401 file: Path, 

402 config: RemoteEngineConfig, 

403 ) -> str | None: 

404 """Send ``file`` to Azure AI Document Intelligence and return text. 

405 

406 Downloads the searchable PDF output from Azure and stores it at 

407 ``self._archive_path``. 

408 

409 Parameters 

410 ---------- 

411 file: 

412 Absolute path to the document to analyse. 

413 config: 

414 Validated remote engine configuration. 

415 

416 Returns 

417 ------- 

418 str | None 

419 Extracted text. 

420 

421 Raises 

422 ------ 

423 ParseError 

424 If the Azure call fails for any reason. The error is logged 

425 and re-raised so consumption fails loudly instead of silently 

426 producing a document with no content. 

427 """ 

428 if TYPE_CHECKING: 

429 # Callers must have already validated config via engine_is_valid(): 

430 # engine_is_valid() asserts api_key is not None and (for azureai) 

431 # endpoint is not None, so these casts are provably safe. 

432 assert config.endpoint is not None 

433 assert config.api_key is not None 

434 

435 from azure.ai.documentintelligence import DocumentIntelligenceClient 

436 from azure.ai.documentintelligence.models import AnalyzeDocumentRequest 

437 from azure.ai.documentintelligence.models import AnalyzeOutputOption 

438 from azure.ai.documentintelligence.models import DocumentContentFormat 

439 from azure.core.credentials import AzureKeyCredential 

440 

441 from paperless.network import validate_outbound_http_url 

442 

443 allow_internal = settings.REMOTE_OCR_ALLOW_INTERNAL_ENDPOINTS 

444 

445 try: 

446 validate_outbound_http_url(config.endpoint, allow_internal=allow_internal) 

447 except ValueError as e: 

448 raise ParseError(f"Invalid remote OCR endpoint: {e}") from e 

449 

450 def _revalidate_request_host(request: PipelineRequest) -> None: 

451 """Re-validates the destination host of every request sent. 

452 

453 The check above only covers the moment the client is built. A 

454 single analysis involves several requests spread over the 

455 polling loop below, and any one of them can be redirected. 

456 Wiring this through ``raw_request_hook`` (Azure's built-in 

457 CustomHookPolicy) rather than a custom policy means it runs 

458 *after* RedirectPolicy in the pipeline, so it sees - and 

459 re-checks - every actual outbound URL, including redirect 

460 targets, not just the original request. 

461 """ 

462 validate_outbound_http_url( 

463 request.http_request.url, 

464 allow_internal=allow_internal, 

465 ) 

466 

467 client = DocumentIntelligenceClient( 

468 endpoint=config.endpoint, 

469 credential=AzureKeyCredential(config.api_key), 

470 raw_request_hook=_revalidate_request_host, 

471 # AzureKeyCredential is sent as Ocp-Apim-Subscription-Key, which 

472 # Azure's default SensitiveHeaderCleanupPolicy does not strip on 

473 # a cross-domain redirect (only Authorization and 

474 # x-ms-authorization-auxiliary are, by default). 

475 blocked_redirect_headers=[ 

476 "Authorization", 

477 "x-ms-authorization-auxiliary", 

478 "Ocp-Apim-Subscription-Key", 

479 ], 

480 ) 

481 

482 try: 

483 with file.open("rb") as f: 

484 analyze_request = AnalyzeDocumentRequest(bytes_source=f.read()) 

485 poller = client.begin_analyze_document( 

486 model_id="prebuilt-read", 

487 body=analyze_request, 

488 output_content_format=DocumentContentFormat.TEXT, 

489 output=[AnalyzeOutputOption.PDF], 

490 content_type="application/json", 

491 ) 

492 

493 poller.wait() 

494 result_id = poller.details["operation_id"] 

495 result = poller.result() 

496 

497 self._archive_path = self._tempdir / "archive.pdf" 

498 with self._archive_path.open("wb") as f: 

499 for chunk in client.get_analyze_result_pdf( 

500 model_id="prebuilt-read", 

501 result_id=result_id, 

502 ): 

503 f.write(chunk) 

504 

505 return result.content 

506 

507 except Exception as e: 

508 logger.exception("Azure AI Vision parsing failed: %s", e) 

509 raise ParseError(f"Azure AI Vision parsing failed: {e}") from e 

510 

511 finally: 

512 client.close()