Coverage for paperless/parsers/__init__.py: 72%
62 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1"""
2Public interface for the Paperless-ngx parser plugin system.
4This module defines ParserProtocol — the structural contract that every
5document parser must satisfy, whether it is a built-in parser shipped with
6Paperless-ngx or a third-party parser installed via a Python entrypoint.
8Phase 1/2 scope: only the Protocol is defined here. The transitional
9DocumentParser ABC (Phase 3) and concrete built-in parsers (Phase 3+) will
10be added in later phases, so there are intentionally no imports of parser
11implementations here.
13Usage example (third-party parser)::
15 from paperless.parsers import ParserProtocol
17 class MyParser:
18 name = "my-parser"
19 version = "1.0.0"
20 author = "Acme Corp"
21 url = "https://example.com/my-parser"
23 @classmethod
24 def supported_mime_types(cls) -> dict[str, str]:
25 return {"application/x-my-format": ".myf"}
27 @classmethod
28 def score(cls, mime_type, filename, path=None):
29 return 10
31 # … implement remaining protocol methods …
33 assert isinstance(MyParser(), ParserProtocol)
34"""
36from __future__ import annotations
38from dataclasses import dataclass
39from typing import TYPE_CHECKING
40from typing import Protocol
41from typing import Self
42from typing import TypedDict
43from typing import runtime_checkable
45if TYPE_CHECKING: 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true
46 import datetime
47 from pathlib import Path
48 from types import TracebackType
50__all__ = [
51 "MetadataEntry",
52 "ParserContext",
53 "ParserProtocol",
54]
57class MetadataEntry(TypedDict):
58 """A single metadata field extracted from a document.
60 All four keys are required. Values are always serialised to strings —
61 type-specific conversion (dates, integers, lists) is the responsibility
62 of the parser before returning.
63 """
65 namespace: str
66 """URI of the metadata namespace (e.g. 'http://ns.adobe.com/pdf/1.3/')."""
68 prefix: str
69 """Conventional namespace prefix (e.g. 'pdf', 'xmp', 'dc')."""
71 key: str
72 """Field name within the namespace (e.g. 'Author', 'CreateDate')."""
74 value: str
75 """String representation of the field value."""
78@dataclass(frozen=True, slots=True)
79class ParserContext:
80 """Immutable context passed to a parser before parse().
82 The consumer assembles this from the ingestion event and Django
83 settings, then calls ``parser.configure(context)`` before
84 ``parser.parse()``. Parsers read only the fields relevant to them;
85 unneeded fields are ignored.
87 ``frozen=True`` prevents accidental mutation after the consumer
88 hands the context off. ``slots=True`` keeps instances lightweight.
90 Fields
91 ------
92 mailrule_id : int | None
93 Primary key of the ``MailRule`` that triggered this ingestion,
94 or ``None`` when the document did not arrive via a mail rule.
95 Used by ``MailDocumentParser`` to select the PDF layout.
97 Notes
98 -----
99 Future fields (not yet implemented):
101 * ``output_type`` — PDF/A variant for archive generation
102 (replaces ``settings.OCR_OUTPUT_TYPE`` reads inside parsers).
103 * ``ocr_mode`` — skip-text, redo, force, etc.
104 (replaces ``settings.OCR_MODE`` reads inside parsers).
105 * ``ocr_language`` — Tesseract language string.
106 (replaces ``settings.OCR_LANGUAGE`` reads inside parsers).
108 When those fields are added the consumer will read from Django
109 settings once and populate them here, decoupling parsers from
110 ``settings.*`` entirely.
111 """
113 mailrule_id: int | None = None
116@runtime_checkable
117class ParserProtocol(Protocol):
118 """Structural contract for all Paperless-ngx document parsers.
120 Both built-in parsers and third-party plugins (discovered via the
121 "paperless_ngx.parsers" entrypoint group) must satisfy this Protocol.
122 Because it is decorated with runtime_checkable, isinstance(obj,
123 ParserProtocol) works at runtime based on method presence, which is
124 useful for validation in ParserRegistry.discover.
126 Parsers must expose four string attributes at the class level so the
127 registry can log attribution information without instantiating the parser:
129 name : str
130 Human-readable parser name (e.g. "Tesseract OCR").
131 version : str
132 Semantic version string (e.g. "1.2.3").
133 author : str
134 Author or organisation name.
135 url : str
136 URL for documentation, source code, or issue tracker.
138 Parsers that send document content to a remote service should additionally
139 set ``uses_remote_service = True`` so the registry can exclude them when
140 remote processing has not been requested for a document. The attribute is
141 optional so a parser that omits it is treated as fully local.
142 """
144 # ------------------------------------------------------------------
145 # Class-level identity (checked by the registry, not Protocol methods)
146 # ------------------------------------------------------------------
148 name: str
149 version: str
150 author: str
151 url: str
153 # NOTE: uses_remote_service is not declared here, the registry reads it
154 # with getattr(cls, ..., False) for backwards-compatibility with existing
155 # parsers
157 # ------------------------------------------------------------------
158 # Class methods
159 # ------------------------------------------------------------------
161 @classmethod
162 def supported_mime_types(cls) -> dict[str, str]:
163 """Return a mapping of supported MIME types to preferred file extensions.
165 The keys are MIME type strings (e.g. "application/pdf"), and the
166 values are the preferred file extension including the leading dot
167 (e.g. ".pdf"). The registry uses this mapping both to decide whether
168 a parser is a candidate for a given file and to determine the default
169 extension when creating archive copies.
171 Returns
172 -------
173 dict[str, str]
174 {mime_type: extension} mapping — may be empty if the parser
175 has been temporarily disabled.
176 """
177 ...
179 @classmethod
180 def score(
181 cls,
182 mime_type: str,
183 filename: str,
184 path: Path | None = None,
185 ) -> int | None:
186 """Return a priority score for handling this file, or None to decline.
188 The registry calls this after confirming that the MIME type is in
189 supported_mime_types. Parsers may inspect filename and optionally
190 the file at path to refine their confidence level.
192 A higher score wins. Return None to explicitly decline handling a file
193 even though the MIME type is listed as supported (e.g. when a feature
194 flag is disabled, or a required service is not configured).
196 Parameters
197 ----------
198 mime_type:
199 The detected MIME type of the file to be parsed.
200 filename:
201 The original filename, including extension.
202 path:
203 Optional filesystem path to the file. Parsers that need to
204 inspect file content (e.g. magic-byte sniffing) may use this.
205 May be None when scoring happens before the file is available locally.
207 Returns
208 -------
209 int | None
210 Priority score (higher wins), or None to decline.
211 """
212 ...
214 # ------------------------------------------------------------------
215 # Properties
216 # ------------------------------------------------------------------
218 @property
219 def can_produce_archive(self) -> bool:
220 """Whether this parser can produce a searchable PDF archive copy.
222 If True, the consumption pipeline may request an archive version when
223 processing the document, subject to the ARCHIVE_FILE_GENERATION
224 setting. If False, only thumbnail and text extraction are performed.
225 """
226 ...
228 @property
229 def requires_pdf_rendition(self) -> bool:
230 """Whether the parser must produce a PDF for the frontend to display.
232 True for formats the browser cannot display natively (e.g. DOCX, ODT).
233 When True, the pipeline always stores the PDF output regardless of the
234 ARCHIVE_FILE_GENERATION setting, since the original format cannot be
235 shown to the user.
236 """
237 ...
239 # ------------------------------------------------------------------
240 # Core parsing interface
241 # ------------------------------------------------------------------
243 def configure(self, context: ParserContext) -> None:
244 """Apply source context before parse().
246 Called by the consumer after instantiation and before parse().
247 The default implementation is a no-op; parsers override only the
248 fields they need.
250 Parameters
251 ----------
252 context:
253 Immutable context assembled by the consumer for this
254 specific ingestion event.
255 """
256 ...
258 def parse(
259 self,
260 document_path: Path,
261 mime_type: str,
262 *,
263 produce_archive: bool = True,
264 ) -> None:
265 """Parse document_path and populate internal state.
267 After a successful call, callers retrieve results via get_text,
268 get_date, and get_archive_path.
270 Parameters
271 ----------
272 document_path:
273 Absolute path to the document file to parse.
274 mime_type:
275 Detected MIME type of the document.
276 produce_archive:
277 When True (the default) and can_produce_archive is also True,
278 the parser should produce a searchable PDF at the path returned
279 by get_archive_path. Pass False when only text extraction and
280 thumbnail generation are required and disk I/O should be minimised.
282 Raises
283 ------
284 documents.parsers.ParseError
285 If parsing fails for any reason.
286 """
287 ...
289 # ------------------------------------------------------------------
290 # Result accessors
291 # ------------------------------------------------------------------
293 def get_text(self) -> str:
294 """Return the plain-text content extracted during parse.
296 Returns
297 -------
298 str
299 Extracted text, or an empty string if no text could be found.
300 """
301 ...
303 def get_date(self) -> datetime.datetime | None:
304 """Return the document date detected during parse.
306 Returns
307 -------
308 datetime.datetime | None
309 Detected document date, or None if no date was found.
310 """
311 ...
313 def get_archive_path(self) -> Path | None:
314 """Return the path to the generated archive PDF, or None.
316 Returns
317 -------
318 Path | None
319 Path to the searchable PDF archive, or None if no archive was
320 produced (e.g. because produce_archive=False or the parser does
321 not support archive generation).
322 """
323 ...
325 # ------------------------------------------------------------------
326 # Thumbnail and metadata
327 # ------------------------------------------------------------------
329 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path:
330 """Generate and return the path to a thumbnail image for the document.
332 May be called independently of parse. The returned path must point to
333 an existing WebP image file inside the parser's temporary working
334 directory.
336 Parameters
337 ----------
338 document_path:
339 Absolute path to the source document.
340 mime_type:
341 Detected MIME type of the document.
343 Returns
344 -------
345 Path
346 Path to the generated thumbnail image (WebP format preferred).
347 """
348 ...
350 def get_page_count(
351 self,
352 document_path: Path,
353 mime_type: str,
354 ) -> int | None:
355 """Return the number of pages in the document, if determinable.
357 Parameters
358 ----------
359 document_path:
360 Absolute path to the source document.
361 mime_type:
362 Detected MIME type of the document.
364 Returns
365 -------
366 int | None
367 Page count, or None if the parser cannot determine it.
368 """
369 ...
371 def extract_metadata(
372 self,
373 document_path: Path,
374 mime_type: str,
375 ) -> list[MetadataEntry]:
376 """Extract format-specific metadata from the document.
378 Called by the API view layer on demand — not during the consumption
379 pipeline. Results are returned to the frontend for per-file display.
381 For documents with an archive version, this method is called twice:
382 once for the original file (with its native MIME type) and once for
383 the archive file (with ``"application/pdf"``). Parsers that produce
384 archives should handle both cases.
386 Implementations must not raise. A failure to read metadata is not
387 fatal — log a warning and return whatever partial results were
388 collected, or ``[]`` if none.
390 Parameters
391 ----------
392 document_path:
393 Absolute path to the file to extract metadata from.
394 mime_type:
395 MIME type of the file at ``document_path``. May be
396 ``"application/pdf"`` when called for the archive version.
398 Returns
399 -------
400 list[MetadataEntry]
401 Zero or more metadata entries. Returns ``[]`` if no metadata
402 could be extracted or the format does not support it.
403 """
404 ...
406 # ------------------------------------------------------------------
407 # Context manager
408 # ------------------------------------------------------------------
410 def __enter__(self) -> Self:
411 """Enter the parser context, returning the parser instance.
413 Implementations should perform any resource allocation here if not
414 done in __init__ (e.g. creating API clients or temp directories).
416 Returns
417 -------
418 Self
419 The parser instance itself.
420 """
421 ...
423 def __exit__(
424 self,
425 exc_type: type[BaseException] | None,
426 exc_val: BaseException | None,
427 exc_tb: TracebackType | None,
428 ) -> None:
429 """Exit the parser context and release all resources.
431 Implementations must clean up all temporary files and other resources
432 regardless of whether an exception occurred.
434 Parameters
435 ----------
436 exc_type:
437 The exception class, or None if no exception was raised.
438 exc_val:
439 The exception instance, or None.
440 exc_tb:
441 The traceback, or None.
442 """
443 ...