Coverage for paperless/parsers/registry.py: 62%

92 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1""" 

2Singleton registry that tracks all document parsers available to 

3Paperless-ngx — both built-ins shipped with the application and third-party 

4plugins installed via Python entrypoints. 

5 

6Public surface 

7-------------- 

8get_parser_registry 

9 Lazy-initialise and return the shared ParserRegistry. This is the primary 

10 entry point for production code. 

11 

12init_builtin_parsers 

13 Register built-in parsers only, without entrypoint discovery. Safe to 

14 call from Celery worker_process_init where importing all entrypoints 

15 would be wasteful or cause side effects. 

16 

17reset_parser_registry 

18 Reset module-level state. For tests only. 

19 

20Entrypoint group 

21---------------- 

22Third-party parsers must advertise themselves under the 

23"paperless_ngx.parsers" entrypoint group in their pyproject.toml:: 

24 

25 [project.entry-points."paperless_ngx.parsers"] 

26 my_parser = "my_package.parsers:MyParser" 

27 

28The loaded class must expose the following attributes at the class level 

29(not just on instances) for the registry to accept it: 

30name, version, author, url, supported_mime_types (callable), score (callable). 

31""" 

32 

33from __future__ import annotations 

34 

35import logging 

36import threading 

37from importlib.metadata import entry_points 

38from typing import TYPE_CHECKING 

39 

40if TYPE_CHECKING: 40 ↛ 41line 40 didn't jump to line 41 because the condition on line 40 was never true

41 from pathlib import Path 

42 

43 from paperless.parsers import ParserProtocol 

44 

45logger = logging.getLogger("paperless.parsers.registry") 

46 

47# --------------------------------------------------------------------------- 

48# Module-level singleton state 

49# --------------------------------------------------------------------------- 

50 

51_registry: ParserRegistry | None = None 

52_discovery_complete: bool = False 

53_lock = threading.Lock() 

54 

55# Attribute names that every registered external parser class must expose. 

56_REQUIRED_ATTRS: tuple[str, ...] = ( 

57 "name", 

58 "version", 

59 "author", 

60 "url", 

61 "supported_mime_types", 

62 "score", 

63) 

64 

65 

66# --------------------------------------------------------------------------- 

67# Module-level accessor functions 

68# --------------------------------------------------------------------------- 

69 

70 

71def get_parser_registry() -> ParserRegistry: 

72 """Return the shared ParserRegistry instance. 

73 

74 On the first call this function: 

75 

76 1. Creates a new ParserRegistry. 

77 2. Calls register_defaults to install built-in parsers. 

78 3. Calls discover to load third-party plugins via importlib.metadata entrypoints. 

79 

80 Subsequent calls return the same instance immediately. 

81 

82 Returns 

83 ------- 

84 ParserRegistry 

85 The shared registry singleton. 

86 """ 

87 global _registry, _discovery_complete 

88 

89 with _lock: 

90 if _registry is None: 

91 r = ParserRegistry() 

92 r.register_defaults() 

93 _registry = r 

94 

95 if not _discovery_complete: 

96 _registry.discover() 

97 _discovery_complete = True 

98 

99 return _registry 

100 

101 

102def init_builtin_parsers() -> None: 

103 """Register built-in parsers without performing entrypoint discovery. 

104 

105 Intended for use in Celery worker_process_init handlers where importing 

106 all installed entrypoints would be wasteful, slow, or could produce 

107 undesirable side effects. Entrypoint discovery (third-party plugins) is 

108 deliberately not performed. 

109 

110 Safe to call multiple times — subsequent calls are no-ops. 

111 

112 Returns 

113 ------- 

114 None 

115 """ 

116 global _registry 

117 

118 with _lock: 

119 if _registry is None: 

120 r = ParserRegistry() 

121 r.register_defaults() 

122 _registry = r 

123 

124 

125def reset_parser_registry() -> None: 

126 """Reset the module-level registry state to its initial values. 

127 

128 Resets _registry and _discovery_complete so the next call to 

129 get_parser_registry will re-initialise everything from scratch. 

130 

131 FOR TESTS ONLY. Do not call this in production code — resetting the 

132 registry mid-request causes all subsequent parser lookups to go through 

133 discovery again, which is expensive and may have unexpected side effects 

134 in multi-threaded environments. 

135 

136 Returns 

137 ------- 

138 None 

139 """ 

140 global _registry, _discovery_complete 

141 

142 _registry = None 

143 _discovery_complete = False 

144 

145 

146# --------------------------------------------------------------------------- 

147# Registry class 

148# --------------------------------------------------------------------------- 

149 

150 

151class ParserRegistry: 

152 """Registry that maps MIME types to the best available parser class. 

153 

154 Parsers are partitioned into two lists: 

155 

156 _builtins 

157 Parser classes registered via register_builtin (populated by 

158 register_defaults in Phase 3+). 

159 

160 _external 

161 Parser classes loaded from installed Python entrypoints via discover. 

162 

163 When resolving a parser for a file, external parsers are evaluated 

164 alongside built-in parsers using a uniform scoring mechanism. Both lists 

165 are iterated together; the class with the highest score wins. If an 

166 external parser wins, its attribution details are logged so users can 

167 identify which third-party package handled their document. 

168 """ 

169 

170 def __init__(self) -> None: 

171 self._external: list[type[ParserProtocol]] = [] 

172 self._builtins: list[type[ParserProtocol]] = [] 

173 

174 # ------------------------------------------------------------------ 

175 # Registration 

176 # ------------------------------------------------------------------ 

177 

178 def register_builtin(self, parser_class: type[ParserProtocol]) -> None: 

179 """Register a built-in parser class. 

180 

181 Built-in parsers are shipped with Paperless-ngx and are appended to 

182 the _builtins list. They are never overridden by external parsers; 

183 instead, scoring determines which parser wins for any given file. 

184 

185 Parameters 

186 ---------- 

187 parser_class: 

188 The parser class to register. Must satisfy ParserProtocol. 

189 """ 

190 self._builtins.append(parser_class) 

191 

192 def register_defaults(self) -> None: 

193 """Register the built-in parsers that ship with Paperless-ngx. 

194 

195 Each parser that has been migrated to the new ParserProtocol interface 

196 is registered here. Parsers are added in ascending weight order so 

197 that log output is predictable; scoring determines which parser wins 

198 at runtime regardless of registration order. 

199 """ 

200 from paperless.parsers.mail import MailDocumentParser 

201 from paperless.parsers.remote import RemoteDocumentParser 

202 from paperless.parsers.tesseract import RasterisedDocumentParser 

203 from paperless.parsers.text import TextDocumentParser 

204 from paperless.parsers.tika import TikaDocumentParser 

205 

206 self.register_builtin(TextDocumentParser) 

207 self.register_builtin(RemoteDocumentParser) 

208 self.register_builtin(TikaDocumentParser) 

209 self.register_builtin(MailDocumentParser) 

210 self.register_builtin(RasterisedDocumentParser) 

211 

212 # ------------------------------------------------------------------ 

213 # Discovery 

214 # ------------------------------------------------------------------ 

215 

216 def discover(self) -> None: 

217 """Load third-party parsers from the "paperless_ngx.parsers" entrypoint group. 

218 

219 For each advertised entrypoint the method: 

220 

221 1. Calls ep.load() to import the class. 

222 2. Validates that the class exposes all required attributes. 

223 3. On success, appends the class to _external and logs an info message. 

224 4. On failure (import error or missing attributes), logs an appropriate 

225 warning/error and continues to the next entrypoint. 

226 

227 Errors during discovery of a single parser do not prevent other parsers 

228 from being loaded. 

229 

230 Returns 

231 ------- 

232 None 

233 """ 

234 eps = entry_points(group="paperless_ngx.parsers") 

235 

236 for ep in eps: 236 ↛ 237line 236 didn't jump to line 237 because the loop on line 236 never started

237 try: 

238 parser_class = ep.load() 

239 except Exception: 

240 logger.exception( 

241 "Failed to load parser entrypoint '%s' — skipping.", 

242 ep.name, 

243 ) 

244 continue 

245 

246 missing = [ 

247 attr for attr in _REQUIRED_ATTRS if not hasattr(parser_class, attr) 

248 ] 

249 if missing: 

250 logger.warning( 

251 "Parser loaded from entrypoint '%s' is missing required " 

252 "attributes %r — skipping.", 

253 ep.name, 

254 missing, 

255 ) 

256 continue 

257 

258 self._external.append(parser_class) 

259 logger.info( 

260 "Loaded third-party parser '%s' v%s by %s (entrypoint: '%s').", 

261 parser_class.name, 

262 parser_class.version, 

263 parser_class.author, 

264 ep.name, 

265 ) 

266 

267 # ------------------------------------------------------------------ 

268 # Summary logging 

269 # ------------------------------------------------------------------ 

270 

271 def log_summary(self) -> None: 

272 """Log a startup summary of all registered parsers. 

273 

274 Built-in parsers are listed first, followed by any external parsers 

275 discovered from entrypoints. If no external parsers were found a 

276 short informational message is logged instead of an empty list. 

277 

278 Returns 

279 ------- 

280 None 

281 """ 

282 logger.info( 

283 "Built-in parsers (%d):", 

284 len(self._builtins), 

285 ) 

286 for cls in self._builtins: 

287 logger.info( 

288 " [built-in] %s v%s — %s", 

289 getattr(cls, "name", repr(cls)), 

290 getattr(cls, "version", "unknown"), 

291 getattr(cls, "url", "built-in"), 

292 ) 

293 

294 if not self._external: 

295 logger.info("No third-party parsers discovered.") 

296 return 

297 

298 logger.info( 

299 "Third-party parsers (%d):", 

300 len(self._external), 

301 ) 

302 for cls in self._external: 

303 logger.info( 

304 " [external] %s v%s by %s — report issues at %s", 

305 getattr(cls, "name", repr(cls)), 

306 getattr(cls, "version", "unknown"), 

307 getattr(cls, "author", "unknown"), 

308 getattr(cls, "url", "unknown"), 

309 ) 

310 

311 # ------------------------------------------------------------------ 

312 # Inspection helpers 

313 # ------------------------------------------------------------------ 

314 

315 def all_parsers(self) -> list[type[ParserProtocol]]: 

316 """Return all registered parser classes (external first, then builtins). 

317 

318 Used by compatibility wrappers that need to iterate every parser to 

319 compute the full set of supported MIME types and file extensions. 

320 

321 Returns 

322 ------- 

323 list[type[ParserProtocol]] 

324 External parsers followed by built-in parsers. 

325 """ 

326 return [*self._external, *self._builtins] 

327 

328 # ------------------------------------------------------------------ 

329 # Parser resolution 

330 # ------------------------------------------------------------------ 

331 

332 def get_parser_for_file( 

333 self, 

334 mime_type: str, 

335 filename: str, 

336 path: Path | None = None, 

337 *, 

338 allow_remote: bool = True, 

339 ) -> type[ParserProtocol] | None: 

340 """Return the best parser class for the given file, or None. 

341 

342 All registered parsers (external first, then built-ins) are evaluated 

343 against the file. A parser is eligible if mime_type appears in the dict 

344 returned by its supported_mime_types classmethod, and its score 

345 classmethod returns a non-None integer. 

346 

347 The parser with the highest score wins. When two parsers return the 

348 same score, the one that appears earlier in the evaluation order wins 

349 (external parsers are evaluated before built-ins, giving third-party 

350 packages a chance to override defaults at equal priority). 

351 

352 When an external parser is selected, its identity is logged at INFO 

353 level so operators can trace which package handled a document. 

354 

355 Parameters 

356 ---------- 

357 mime_type: 

358 The detected MIME type of the file. 

359 filename: 

360 The original filename, including extension. May be empty in some cases 

361 path: 

362 Optional filesystem path to the file. Forwarded to each 

363 parser's score method. 

364 allow_remote: 

365 When False, parsers that declare ``uses_remote_service = True`` 

366 are excluded from consideration, so a document is never sent to 

367 a remote service. Parsers that do not declare the attribute 

368 are treated as local and are always considered. 

369 

370 Returns 

371 ------- 

372 type[ParserProtocol] | None 

373 The winning parser class, or None if no parser can handle the file. 

374 """ 

375 best_score: int | None = None 

376 best_parser: type[ParserProtocol] | None = None 

377 

378 # External parsers are placed first so that, at equal scores, an 

379 # external parser wins over a built-in (first-seen policy). 

380 for parser_class in (*self._external, *self._builtins): 

381 if mime_type not in parser_class.supported_mime_types(): 

382 continue 

383 

384 if not allow_remote and getattr( 384 ↛ 389line 384 didn't jump to line 389 because the condition on line 384 was never true

385 parser_class, 

386 "uses_remote_service", 

387 False, 

388 ): 

389 continue 

390 

391 score = parser_class.score(mime_type, filename, path) 

392 if score is None: 

393 continue 

394 

395 if best_score is None or score > best_score: 395 ↛ 380line 395 didn't jump to line 380 because the condition on line 395 was always true

396 best_score = score 

397 best_parser = parser_class 

398 

399 if best_parser is not None and best_parser in self._external: 399 ↛ 400line 399 didn't jump to line 400 because the condition on line 399 was never true

400 logger.info( 

401 "Document handled by third-party parser '%s' v%s — %s", 

402 getattr(best_parser, "name", repr(best_parser)), 

403 getattr(best_parser, "version", "unknown"), 

404 getattr(best_parser, "url", "unknown"), 

405 ) 

406 

407 return best_parser