Coverage for paperless/parsers/registry.py: 62%
92 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1"""
2Singleton registry that tracks all document parsers available to
3Paperless-ngx — both built-ins shipped with the application and third-party
4plugins installed via Python entrypoints.
6Public surface
7--------------
8get_parser_registry
9 Lazy-initialise and return the shared ParserRegistry. This is the primary
10 entry point for production code.
12init_builtin_parsers
13 Register built-in parsers only, without entrypoint discovery. Safe to
14 call from Celery worker_process_init where importing all entrypoints
15 would be wasteful or cause side effects.
17reset_parser_registry
18 Reset module-level state. For tests only.
20Entrypoint group
21----------------
22Third-party parsers must advertise themselves under the
23"paperless_ngx.parsers" entrypoint group in their pyproject.toml::
25 [project.entry-points."paperless_ngx.parsers"]
26 my_parser = "my_package.parsers:MyParser"
28The loaded class must expose the following attributes at the class level
29(not just on instances) for the registry to accept it:
30name, version, author, url, supported_mime_types (callable), score (callable).
31"""
33from __future__ import annotations
35import logging
36import threading
37from importlib.metadata import entry_points
38from typing import TYPE_CHECKING
40if TYPE_CHECKING: 40 ↛ 41line 40 didn't jump to line 41 because the condition on line 40 was never true
41 from pathlib import Path
43 from paperless.parsers import ParserProtocol
45logger = logging.getLogger("paperless.parsers.registry")
47# ---------------------------------------------------------------------------
48# Module-level singleton state
49# ---------------------------------------------------------------------------
51_registry: ParserRegistry | None = None
52_discovery_complete: bool = False
53_lock = threading.Lock()
55# Attribute names that every registered external parser class must expose.
56_REQUIRED_ATTRS: tuple[str, ...] = (
57 "name",
58 "version",
59 "author",
60 "url",
61 "supported_mime_types",
62 "score",
63)
66# ---------------------------------------------------------------------------
67# Module-level accessor functions
68# ---------------------------------------------------------------------------
71def get_parser_registry() -> ParserRegistry:
72 """Return the shared ParserRegistry instance.
74 On the first call this function:
76 1. Creates a new ParserRegistry.
77 2. Calls register_defaults to install built-in parsers.
78 3. Calls discover to load third-party plugins via importlib.metadata entrypoints.
80 Subsequent calls return the same instance immediately.
82 Returns
83 -------
84 ParserRegistry
85 The shared registry singleton.
86 """
87 global _registry, _discovery_complete
89 with _lock:
90 if _registry is None:
91 r = ParserRegistry()
92 r.register_defaults()
93 _registry = r
95 if not _discovery_complete:
96 _registry.discover()
97 _discovery_complete = True
99 return _registry
102def init_builtin_parsers() -> None:
103 """Register built-in parsers without performing entrypoint discovery.
105 Intended for use in Celery worker_process_init handlers where importing
106 all installed entrypoints would be wasteful, slow, or could produce
107 undesirable side effects. Entrypoint discovery (third-party plugins) is
108 deliberately not performed.
110 Safe to call multiple times — subsequent calls are no-ops.
112 Returns
113 -------
114 None
115 """
116 global _registry
118 with _lock:
119 if _registry is None:
120 r = ParserRegistry()
121 r.register_defaults()
122 _registry = r
125def reset_parser_registry() -> None:
126 """Reset the module-level registry state to its initial values.
128 Resets _registry and _discovery_complete so the next call to
129 get_parser_registry will re-initialise everything from scratch.
131 FOR TESTS ONLY. Do not call this in production code — resetting the
132 registry mid-request causes all subsequent parser lookups to go through
133 discovery again, which is expensive and may have unexpected side effects
134 in multi-threaded environments.
136 Returns
137 -------
138 None
139 """
140 global _registry, _discovery_complete
142 _registry = None
143 _discovery_complete = False
146# ---------------------------------------------------------------------------
147# Registry class
148# ---------------------------------------------------------------------------
151class ParserRegistry:
152 """Registry that maps MIME types to the best available parser class.
154 Parsers are partitioned into two lists:
156 _builtins
157 Parser classes registered via register_builtin (populated by
158 register_defaults in Phase 3+).
160 _external
161 Parser classes loaded from installed Python entrypoints via discover.
163 When resolving a parser for a file, external parsers are evaluated
164 alongside built-in parsers using a uniform scoring mechanism. Both lists
165 are iterated together; the class with the highest score wins. If an
166 external parser wins, its attribution details are logged so users can
167 identify which third-party package handled their document.
168 """
170 def __init__(self) -> None:
171 self._external: list[type[ParserProtocol]] = []
172 self._builtins: list[type[ParserProtocol]] = []
174 # ------------------------------------------------------------------
175 # Registration
176 # ------------------------------------------------------------------
178 def register_builtin(self, parser_class: type[ParserProtocol]) -> None:
179 """Register a built-in parser class.
181 Built-in parsers are shipped with Paperless-ngx and are appended to
182 the _builtins list. They are never overridden by external parsers;
183 instead, scoring determines which parser wins for any given file.
185 Parameters
186 ----------
187 parser_class:
188 The parser class to register. Must satisfy ParserProtocol.
189 """
190 self._builtins.append(parser_class)
192 def register_defaults(self) -> None:
193 """Register the built-in parsers that ship with Paperless-ngx.
195 Each parser that has been migrated to the new ParserProtocol interface
196 is registered here. Parsers are added in ascending weight order so
197 that log output is predictable; scoring determines which parser wins
198 at runtime regardless of registration order.
199 """
200 from paperless.parsers.mail import MailDocumentParser
201 from paperless.parsers.remote import RemoteDocumentParser
202 from paperless.parsers.tesseract import RasterisedDocumentParser
203 from paperless.parsers.text import TextDocumentParser
204 from paperless.parsers.tika import TikaDocumentParser
206 self.register_builtin(TextDocumentParser)
207 self.register_builtin(RemoteDocumentParser)
208 self.register_builtin(TikaDocumentParser)
209 self.register_builtin(MailDocumentParser)
210 self.register_builtin(RasterisedDocumentParser)
212 # ------------------------------------------------------------------
213 # Discovery
214 # ------------------------------------------------------------------
216 def discover(self) -> None:
217 """Load third-party parsers from the "paperless_ngx.parsers" entrypoint group.
219 For each advertised entrypoint the method:
221 1. Calls ep.load() to import the class.
222 2. Validates that the class exposes all required attributes.
223 3. On success, appends the class to _external and logs an info message.
224 4. On failure (import error or missing attributes), logs an appropriate
225 warning/error and continues to the next entrypoint.
227 Errors during discovery of a single parser do not prevent other parsers
228 from being loaded.
230 Returns
231 -------
232 None
233 """
234 eps = entry_points(group="paperless_ngx.parsers")
236 for ep in eps: 236 ↛ 237line 236 didn't jump to line 237 because the loop on line 236 never started
237 try:
238 parser_class = ep.load()
239 except Exception:
240 logger.exception(
241 "Failed to load parser entrypoint '%s' — skipping.",
242 ep.name,
243 )
244 continue
246 missing = [
247 attr for attr in _REQUIRED_ATTRS if not hasattr(parser_class, attr)
248 ]
249 if missing:
250 logger.warning(
251 "Parser loaded from entrypoint '%s' is missing required "
252 "attributes %r — skipping.",
253 ep.name,
254 missing,
255 )
256 continue
258 self._external.append(parser_class)
259 logger.info(
260 "Loaded third-party parser '%s' v%s by %s (entrypoint: '%s').",
261 parser_class.name,
262 parser_class.version,
263 parser_class.author,
264 ep.name,
265 )
267 # ------------------------------------------------------------------
268 # Summary logging
269 # ------------------------------------------------------------------
271 def log_summary(self) -> None:
272 """Log a startup summary of all registered parsers.
274 Built-in parsers are listed first, followed by any external parsers
275 discovered from entrypoints. If no external parsers were found a
276 short informational message is logged instead of an empty list.
278 Returns
279 -------
280 None
281 """
282 logger.info(
283 "Built-in parsers (%d):",
284 len(self._builtins),
285 )
286 for cls in self._builtins:
287 logger.info(
288 " [built-in] %s v%s — %s",
289 getattr(cls, "name", repr(cls)),
290 getattr(cls, "version", "unknown"),
291 getattr(cls, "url", "built-in"),
292 )
294 if not self._external:
295 logger.info("No third-party parsers discovered.")
296 return
298 logger.info(
299 "Third-party parsers (%d):",
300 len(self._external),
301 )
302 for cls in self._external:
303 logger.info(
304 " [external] %s v%s by %s — report issues at %s",
305 getattr(cls, "name", repr(cls)),
306 getattr(cls, "version", "unknown"),
307 getattr(cls, "author", "unknown"),
308 getattr(cls, "url", "unknown"),
309 )
311 # ------------------------------------------------------------------
312 # Inspection helpers
313 # ------------------------------------------------------------------
315 def all_parsers(self) -> list[type[ParserProtocol]]:
316 """Return all registered parser classes (external first, then builtins).
318 Used by compatibility wrappers that need to iterate every parser to
319 compute the full set of supported MIME types and file extensions.
321 Returns
322 -------
323 list[type[ParserProtocol]]
324 External parsers followed by built-in parsers.
325 """
326 return [*self._external, *self._builtins]
328 # ------------------------------------------------------------------
329 # Parser resolution
330 # ------------------------------------------------------------------
332 def get_parser_for_file(
333 self,
334 mime_type: str,
335 filename: str,
336 path: Path | None = None,
337 *,
338 allow_remote: bool = True,
339 ) -> type[ParserProtocol] | None:
340 """Return the best parser class for the given file, or None.
342 All registered parsers (external first, then built-ins) are evaluated
343 against the file. A parser is eligible if mime_type appears in the dict
344 returned by its supported_mime_types classmethod, and its score
345 classmethod returns a non-None integer.
347 The parser with the highest score wins. When two parsers return the
348 same score, the one that appears earlier in the evaluation order wins
349 (external parsers are evaluated before built-ins, giving third-party
350 packages a chance to override defaults at equal priority).
352 When an external parser is selected, its identity is logged at INFO
353 level so operators can trace which package handled a document.
355 Parameters
356 ----------
357 mime_type:
358 The detected MIME type of the file.
359 filename:
360 The original filename, including extension. May be empty in some cases
361 path:
362 Optional filesystem path to the file. Forwarded to each
363 parser's score method.
364 allow_remote:
365 When False, parsers that declare ``uses_remote_service = True``
366 are excluded from consideration, so a document is never sent to
367 a remote service. Parsers that do not declare the attribute
368 are treated as local and are always considered.
370 Returns
371 -------
372 type[ParserProtocol] | None
373 The winning parser class, or None if no parser can handle the file.
374 """
375 best_score: int | None = None
376 best_parser: type[ParserProtocol] | None = None
378 # External parsers are placed first so that, at equal scores, an
379 # external parser wins over a built-in (first-seen policy).
380 for parser_class in (*self._external, *self._builtins):
381 if mime_type not in parser_class.supported_mime_types():
382 continue
384 if not allow_remote and getattr( 384 ↛ 389line 384 didn't jump to line 389 because the condition on line 384 was never true
385 parser_class,
386 "uses_remote_service",
387 False,
388 ):
389 continue
391 score = parser_class.score(mime_type, filename, path)
392 if score is None:
393 continue
395 if best_score is None or score > best_score: 395 ↛ 380line 395 didn't jump to line 380 because the condition on line 395 was always true
396 best_score = score
397 best_parser = parser_class
399 if best_parser is not None and best_parser in self._external: 399 ↛ 400line 399 didn't jump to line 400 because the condition on line 399 was never true
400 logger.info(
401 "Document handled by third-party parser '%s' v%s — %s",
402 getattr(best_parser, "name", repr(best_parser)),
403 getattr(best_parser, "version", "unknown"),
404 getattr(best_parser, "url", "unknown"),
405 )
407 return best_parser