Coverage for paperless/parsers/remote.py: 39%
133 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1"""
2Built-in remote-OCR document parser.
4Handles documents by sending them to a configured remote OCR engine
5(currently Azure AI Vision / Document Intelligence) and retrieving both
6the extracted text and a searchable PDF with an embedded text layer. For
7born-digital PDFs that need no archive copy, the remote call is skipped
8entirely in favor of locally-extracted text (see ``RemoteDocumentParser.parse``).
10When no engine is configured, ``score()`` returns ``None`` so the parser
11is effectively invisible to the registry — the tesseract parser handles
12these MIME types instead.
13"""
15from __future__ import annotations
17import logging
18import shutil
19import tempfile
20from pathlib import Path
21from typing import TYPE_CHECKING
22from typing import Self
24from django.conf import settings
26from documents.parsers import ParseError
27from paperless.parsers.utils import extract_pdf_text
28from paperless.parsers.utils import post_process_text
29from paperless.version import __full_version_str__
31if TYPE_CHECKING: 31 ↛ 32line 31 didn't jump to line 32 because the condition on line 31 was never true
32 import datetime
33 from types import TracebackType
35 from azure.core.pipeline import PipelineRequest
37 from paperless.parsers import MetadataEntry
38 from paperless.parsers import ParserContext
40logger = logging.getLogger("paperless.parsing.remote")
42_SUPPORTED_MIME_TYPES: dict[str, str] = {
43 "application/pdf": ".pdf",
44 "image/png": ".png",
45 "image/jpeg": ".jpg",
46 "image/tiff": ".tiff",
47 "image/bmp": ".bmp",
48 "image/gif": ".gif",
49 "image/webp": ".webp",
50}
53class RemoteEngineConfig:
54 """Holds and validates the remote OCR engine configuration."""
56 def __init__(
57 self,
58 engine: str | None,
59 api_key: str | None = None,
60 endpoint: str | None = None,
61 ) -> None:
62 self.engine = engine
63 self.api_key = api_key
64 self.endpoint = endpoint
66 @classmethod
67 def from_app_config(cls) -> Self:
68 """Build the config from the app config, falling back to the env."""
69 from paperless.config import RemoteOCRConfig
71 app_config = RemoteOCRConfig()
72 return cls(
73 engine=app_config.remote_ocr_engine,
74 api_key=app_config.remote_ocr_api_key,
75 endpoint=app_config.remote_ocr_endpoint,
76 )
78 def engine_is_valid(self) -> bool:
79 """Return True when the engine is known and fully configured."""
80 return (
81 self.engine in ("azureai",)
82 and self.api_key is not None
83 and not (self.engine == "azureai" and self.endpoint is None)
84 )
87class RemoteDocumentParser:
88 """Parse documents via a remote OCR API (currently Azure AI Vision).
90 This parser sends documents to a remote engine that returns both
91 extracted text and a searchable PDF with an embedded text layer,
92 except when ``parse()`` is called with ``produce_archive=False`` for
93 a PDF, in which case the remote call is skipped and only locally
94 extracted text is returned (no archive). It does not depend on
95 Tesseract or ocrmypdf.
97 Class attributes
98 ----------------
99 name : str
100 Human-readable parser name.
101 version : str
102 Semantic version string, kept in sync with Paperless-ngx releases.
103 author : str
104 Maintainer name.
105 url : str
106 Issue tracker / source URL.
107 uses_remote_service : bool
108 Content is sent to a remote service, True so that the registry
109 can skip this parser if remote processing was not requested.
110 """
112 name: str = "Paperless-ngx Remote OCR Parser"
113 version: str = __full_version_str__
114 author: str = "Paperless-ngx Contributors"
115 url: str = "https://github.com/paperless-ngx/paperless-ngx"
117 uses_remote_service: bool = True
119 # ------------------------------------------------------------------
120 # Class methods
121 # ------------------------------------------------------------------
123 @classmethod
124 def supported_mime_types(cls) -> dict[str, str]:
125 """Return the MIME types this parser can handle.
127 The full set is always returned regardless of whether a remote
128 engine is configured. The ``score()`` method handles the
129 "am I active?" logic by returning ``None`` when not configured.
131 Returns
132 -------
133 dict[str, str]
134 Mapping of MIME type to preferred file extension.
135 """
136 return _SUPPORTED_MIME_TYPES
138 @classmethod
139 def score(
140 cls,
141 mime_type: str,
142 filename: str,
143 path: Path | None = None,
144 ) -> int | None:
145 """Return the priority score for handling this file, or None.
147 Returns ``None`` when no valid remote engine is configured,
148 making the parser invisible to the registry for this file.
149 When configured, returns 20 — higher than the Tesseract parser's
150 default of 10 — so the remote engine takes priority.
152 Parameters
153 ----------
154 mime_type:
155 Detected MIME type of the file.
156 filename:
157 Original filename including extension.
158 path:
159 Optional filesystem path. Not inspected by this parser.
161 Returns
162 -------
163 int | None
164 20 when the remote engine is configured and the MIME type is
165 supported, otherwise None.
166 """
167 config = RemoteEngineConfig.from_app_config()
168 if not config.engine_is_valid(): 168 ↛ 170line 168 didn't jump to line 170 because the condition on line 168 was always true
169 return None
170 if mime_type not in _SUPPORTED_MIME_TYPES:
171 return None
172 return 20
174 # ------------------------------------------------------------------
175 # Properties
176 # ------------------------------------------------------------------
178 @property
179 def can_produce_archive(self) -> bool:
180 """Whether this parser can produce a searchable PDF archive copy.
182 Returns
183 -------
184 bool
185 Always True — the remote engine is capable of returning a PDF
186 with an embedded text layer to serve as the archive copy.
187 Whether it actually does so for a given document depends on
188 ``produce_archive`` passed to :meth:`parse` (see there for when
189 the remote engine call, and thus archive generation, is skipped).
190 """
191 return True
193 @property
194 def requires_pdf_rendition(self) -> bool:
195 """Whether the parser must produce a PDF for the frontend to display.
197 Returns
198 -------
199 bool
200 Always False — all supported originals are displayable by
201 the browser (PDF) or handled via the archive copy (images).
202 """
203 return False
205 # ------------------------------------------------------------------
206 # Lifecycle
207 # ------------------------------------------------------------------
209 def __init__(self, logging_group: object = None) -> None:
210 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True)
211 self._tempdir = Path(
212 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR),
213 )
214 self._logging_group = logging_group
215 self._text: str | None = None
216 self._archive_path: Path | None = None
218 def __enter__(self) -> Self:
219 return self
221 def __exit__(
222 self,
223 exc_type: type[BaseException] | None,
224 exc_val: BaseException | None,
225 exc_tb: TracebackType | None,
226 ) -> None:
227 logger.debug("Cleaning up temporary directory %s", self._tempdir)
228 shutil.rmtree(self._tempdir, ignore_errors=True)
230 # ------------------------------------------------------------------
231 # Core parsing interface
232 # ------------------------------------------------------------------
234 def configure(self, context: ParserContext) -> None:
235 pass
237 def parse(
238 self,
239 document_path: Path,
240 mime_type: str,
241 *,
242 produce_archive: bool = True,
243 ) -> None:
244 """Send the document to the remote engine and store results.
246 When *produce_archive* is False for a PDF, the caller (via
247 ``documents.consumer.should_produce_archive``) has already determined
248 that the document is born-digital and needs no archive — skip the
249 remote engine entirely rather than re-OCRing it and creating a
250 duplicate text layer.
252 Parameters
253 ----------
254 document_path:
255 Absolute path to the document file to parse.
256 mime_type:
257 Detected MIME type of the document.
258 produce_archive:
259 Whether an archive copy is wanted. For PDFs, False skips the
260 remote engine and uses locally-extracted text instead.
261 """
262 config = RemoteEngineConfig.from_app_config()
264 if not config.engine_is_valid():
265 logger.warning(
266 "No valid remote parser engine is configured, content will be empty.",
267 )
268 self._text = ""
269 return
271 if not produce_archive and mime_type == "application/pdf":
272 logger.debug(
273 "Remote OCR: skipped — no archive requested, "
274 "using locally-extracted text",
275 )
276 self._text = (
277 post_process_text(extract_pdf_text(document_path, log=logger)) or ""
278 )
279 return
281 if config.engine == "azureai":
282 self._text = self._azure_ai_vision_parse(document_path, config)
284 # ------------------------------------------------------------------
285 # Result accessors
286 # ------------------------------------------------------------------
288 def get_text(self) -> str:
289 """Return the plain-text content extracted during parse."""
290 return self._text or ""
292 def get_date(self) -> datetime.datetime | None:
293 """Return the document date detected during parse.
295 Returns
296 -------
297 datetime.datetime | None
298 Always None — the remote parser does not detect dates.
299 """
300 return None
302 def get_archive_path(self) -> Path | None:
303 """Return the path to the generated archive PDF, or None."""
304 return self._archive_path
306 # ------------------------------------------------------------------
307 # Thumbnail and metadata
308 # ------------------------------------------------------------------
310 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path:
311 """Generate a thumbnail image for the document.
313 Uses the archive PDF produced by the remote engine when available,
314 otherwise falls back to the original document path (PDF inputs).
316 Parameters
317 ----------
318 document_path:
319 Absolute path to the source document.
320 mime_type:
321 Detected MIME type of the document.
323 Returns
324 -------
325 Path
326 Path to the generated WebP thumbnail inside the temp directory.
327 """
328 # make_thumbnail_from_pdf lives in documents.parsers for now;
329 # it will move to paperless.parsers.utils when the tesseract
330 # parser is migrated in a later phase.
331 from documents.parsers import make_thumbnail_from_pdf
333 return make_thumbnail_from_pdf(
334 self._archive_path or document_path,
335 self._tempdir,
336 self._logging_group,
337 )
339 def get_page_count(
340 self,
341 document_path: Path,
342 mime_type: str,
343 ) -> int | None:
344 """Return the number of pages in a PDF document.
346 Parameters
347 ----------
348 document_path:
349 Absolute path to the source document.
350 mime_type:
351 Detected MIME type of the document.
353 Returns
354 -------
355 int | None
356 Page count for PDF inputs, or ``None`` for other MIME types.
357 """
358 if mime_type != "application/pdf":
359 return None
361 from paperless.parsers.utils import get_page_count_for_pdf
363 return get_page_count_for_pdf(document_path, log=logger)
365 def extract_metadata(
366 self,
367 document_path: Path,
368 mime_type: str,
369 ) -> list[MetadataEntry]:
370 """Extract format-specific metadata from the document.
372 Delegates to the shared pikepdf-based extractor for PDF files.
373 Returns ``[]`` for all other MIME types.
375 Parameters
376 ----------
377 document_path:
378 Absolute path to the file to extract metadata from.
379 mime_type:
380 MIME type of the file. May be ``"application/pdf"`` when
381 called for the archive version of an image original.
383 Returns
384 -------
385 list[MetadataEntry]
386 Zero or more metadata entries.
387 """
388 if mime_type != "application/pdf":
389 return []
391 from paperless.parsers.utils import extract_pdf_metadata
393 return extract_pdf_metadata(document_path, log=logger)
395 # ------------------------------------------------------------------
396 # Private helpers
397 # ------------------------------------------------------------------
399 def _azure_ai_vision_parse(
400 self,
401 file: Path,
402 config: RemoteEngineConfig,
403 ) -> str | None:
404 """Send ``file`` to Azure AI Document Intelligence and return text.
406 Downloads the searchable PDF output from Azure and stores it at
407 ``self._archive_path``.
409 Parameters
410 ----------
411 file:
412 Absolute path to the document to analyse.
413 config:
414 Validated remote engine configuration.
416 Returns
417 -------
418 str | None
419 Extracted text.
421 Raises
422 ------
423 ParseError
424 If the Azure call fails for any reason. The error is logged
425 and re-raised so consumption fails loudly instead of silently
426 producing a document with no content.
427 """
428 if TYPE_CHECKING:
429 # Callers must have already validated config via engine_is_valid():
430 # engine_is_valid() asserts api_key is not None and (for azureai)
431 # endpoint is not None, so these casts are provably safe.
432 assert config.endpoint is not None
433 assert config.api_key is not None
435 from azure.ai.documentintelligence import DocumentIntelligenceClient
436 from azure.ai.documentintelligence.models import AnalyzeDocumentRequest
437 from azure.ai.documentintelligence.models import AnalyzeOutputOption
438 from azure.ai.documentintelligence.models import DocumentContentFormat
439 from azure.core.credentials import AzureKeyCredential
441 from paperless.network import validate_outbound_http_url
443 allow_internal = settings.REMOTE_OCR_ALLOW_INTERNAL_ENDPOINTS
445 try:
446 validate_outbound_http_url(config.endpoint, allow_internal=allow_internal)
447 except ValueError as e:
448 raise ParseError(f"Invalid remote OCR endpoint: {e}") from e
450 def _revalidate_request_host(request: PipelineRequest) -> None:
451 """Re-validates the destination host of every request sent.
453 The check above only covers the moment the client is built. A
454 single analysis involves several requests spread over the
455 polling loop below, and any one of them can be redirected.
456 Wiring this through ``raw_request_hook`` (Azure's built-in
457 CustomHookPolicy) rather than a custom policy means it runs
458 *after* RedirectPolicy in the pipeline, so it sees - and
459 re-checks - every actual outbound URL, including redirect
460 targets, not just the original request.
461 """
462 validate_outbound_http_url(
463 request.http_request.url,
464 allow_internal=allow_internal,
465 )
467 client = DocumentIntelligenceClient(
468 endpoint=config.endpoint,
469 credential=AzureKeyCredential(config.api_key),
470 raw_request_hook=_revalidate_request_host,
471 # AzureKeyCredential is sent as Ocp-Apim-Subscription-Key, which
472 # Azure's default SensitiveHeaderCleanupPolicy does not strip on
473 # a cross-domain redirect (only Authorization and
474 # x-ms-authorization-auxiliary are, by default).
475 blocked_redirect_headers=[
476 "Authorization",
477 "x-ms-authorization-auxiliary",
478 "Ocp-Apim-Subscription-Key",
479 ],
480 )
482 try:
483 with file.open("rb") as f:
484 analyze_request = AnalyzeDocumentRequest(bytes_source=f.read())
485 poller = client.begin_analyze_document(
486 model_id="prebuilt-read",
487 body=analyze_request,
488 output_content_format=DocumentContentFormat.TEXT,
489 output=[AnalyzeOutputOption.PDF],
490 content_type="application/json",
491 )
493 poller.wait()
494 result_id = poller.details["operation_id"]
495 result = poller.result()
497 self._archive_path = self._tempdir / "archive.pdf"
498 with self._archive_path.open("wb") as f:
499 for chunk in client.get_analyze_result_pdf(
500 model_id="prebuilt-read",
501 result_id=result_id,
502 ):
503 f.write(chunk)
505 return result.content
507 except Exception as e:
508 logger.exception("Azure AI Vision parsing failed: %s", e)
509 raise ParseError(f"Azure AI Vision parsing failed: {e}") from e
511 finally:
512 client.close()