Coverage for paperless/parsers/__init__.py: 72%

62 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1""" 

2Public interface for the Paperless-ngx parser plugin system. 

3 

4This module defines ParserProtocol — the structural contract that every 

5document parser must satisfy, whether it is a built-in parser shipped with 

6Paperless-ngx or a third-party parser installed via a Python entrypoint. 

7 

8Phase 1/2 scope: only the Protocol is defined here. The transitional 

9DocumentParser ABC (Phase 3) and concrete built-in parsers (Phase 3+) will 

10be added in later phases, so there are intentionally no imports of parser 

11implementations here. 

12 

13Usage example (third-party parser):: 

14 

15 from paperless.parsers import ParserProtocol 

16 

17 class MyParser: 

18 name = "my-parser" 

19 version = "1.0.0" 

20 author = "Acme Corp" 

21 url = "https://example.com/my-parser" 

22 

23 @classmethod 

24 def supported_mime_types(cls) -> dict[str, str]: 

25 return {"application/x-my-format": ".myf"} 

26 

27 @classmethod 

28 def score(cls, mime_type, filename, path=None): 

29 return 10 

30 

31 # … implement remaining protocol methods … 

32 

33 assert isinstance(MyParser(), ParserProtocol) 

34""" 

35 

36from __future__ import annotations 

37 

38from dataclasses import dataclass 

39from typing import TYPE_CHECKING 

40from typing import Protocol 

41from typing import Self 

42from typing import TypedDict 

43from typing import runtime_checkable 

44 

45if TYPE_CHECKING: 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true

46 import datetime 

47 from pathlib import Path 

48 from types import TracebackType 

49 

50__all__ = [ 

51 "MetadataEntry", 

52 "ParserContext", 

53 "ParserProtocol", 

54] 

55 

56 

57class MetadataEntry(TypedDict): 

58 """A single metadata field extracted from a document. 

59 

60 All four keys are required. Values are always serialised to strings — 

61 type-specific conversion (dates, integers, lists) is the responsibility 

62 of the parser before returning. 

63 """ 

64 

65 namespace: str 

66 """URI of the metadata namespace (e.g. 'http://ns.adobe.com/pdf/1.3/').""" 

67 

68 prefix: str 

69 """Conventional namespace prefix (e.g. 'pdf', 'xmp', 'dc').""" 

70 

71 key: str 

72 """Field name within the namespace (e.g. 'Author', 'CreateDate').""" 

73 

74 value: str 

75 """String representation of the field value.""" 

76 

77 

78@dataclass(frozen=True, slots=True) 

79class ParserContext: 

80 """Immutable context passed to a parser before parse(). 

81 

82 The consumer assembles this from the ingestion event and Django 

83 settings, then calls ``parser.configure(context)`` before 

84 ``parser.parse()``. Parsers read only the fields relevant to them; 

85 unneeded fields are ignored. 

86 

87 ``frozen=True`` prevents accidental mutation after the consumer 

88 hands the context off. ``slots=True`` keeps instances lightweight. 

89 

90 Fields 

91 ------ 

92 mailrule_id : int | None 

93 Primary key of the ``MailRule`` that triggered this ingestion, 

94 or ``None`` when the document did not arrive via a mail rule. 

95 Used by ``MailDocumentParser`` to select the PDF layout. 

96 

97 Notes 

98 ----- 

99 Future fields (not yet implemented): 

100 

101 * ``output_type`` — PDF/A variant for archive generation 

102 (replaces ``settings.OCR_OUTPUT_TYPE`` reads inside parsers). 

103 * ``ocr_mode`` — skip-text, redo, force, etc. 

104 (replaces ``settings.OCR_MODE`` reads inside parsers). 

105 * ``ocr_language`` — Tesseract language string. 

106 (replaces ``settings.OCR_LANGUAGE`` reads inside parsers). 

107 

108 When those fields are added the consumer will read from Django 

109 settings once and populate them here, decoupling parsers from 

110 ``settings.*`` entirely. 

111 """ 

112 

113 mailrule_id: int | None = None 

114 

115 

116@runtime_checkable 

117class ParserProtocol(Protocol): 

118 """Structural contract for all Paperless-ngx document parsers. 

119 

120 Both built-in parsers and third-party plugins (discovered via the 

121 "paperless_ngx.parsers" entrypoint group) must satisfy this Protocol. 

122 Because it is decorated with runtime_checkable, isinstance(obj, 

123 ParserProtocol) works at runtime based on method presence, which is 

124 useful for validation in ParserRegistry.discover. 

125 

126 Parsers must expose four string attributes at the class level so the 

127 registry can log attribution information without instantiating the parser: 

128 

129 name : str 

130 Human-readable parser name (e.g. "Tesseract OCR"). 

131 version : str 

132 Semantic version string (e.g. "1.2.3"). 

133 author : str 

134 Author or organisation name. 

135 url : str 

136 URL for documentation, source code, or issue tracker. 

137 

138 Parsers that send document content to a remote service should additionally 

139 set ``uses_remote_service = True`` so the registry can exclude them when 

140 remote processing has not been requested for a document. The attribute is 

141 optional so a parser that omits it is treated as fully local. 

142 """ 

143 

144 # ------------------------------------------------------------------ 

145 # Class-level identity (checked by the registry, not Protocol methods) 

146 # ------------------------------------------------------------------ 

147 

148 name: str 

149 version: str 

150 author: str 

151 url: str 

152 

153 # NOTE: uses_remote_service is not declared here, the registry reads it 

154 # with getattr(cls, ..., False) for backwards-compatibility with existing 

155 # parsers 

156 

157 # ------------------------------------------------------------------ 

158 # Class methods 

159 # ------------------------------------------------------------------ 

160 

161 @classmethod 

162 def supported_mime_types(cls) -> dict[str, str]: 

163 """Return a mapping of supported MIME types to preferred file extensions. 

164 

165 The keys are MIME type strings (e.g. "application/pdf"), and the 

166 values are the preferred file extension including the leading dot 

167 (e.g. ".pdf"). The registry uses this mapping both to decide whether 

168 a parser is a candidate for a given file and to determine the default 

169 extension when creating archive copies. 

170 

171 Returns 

172 ------- 

173 dict[str, str] 

174 {mime_type: extension} mapping — may be empty if the parser 

175 has been temporarily disabled. 

176 """ 

177 ... 

178 

179 @classmethod 

180 def score( 

181 cls, 

182 mime_type: str, 

183 filename: str, 

184 path: Path | None = None, 

185 ) -> int | None: 

186 """Return a priority score for handling this file, or None to decline. 

187 

188 The registry calls this after confirming that the MIME type is in 

189 supported_mime_types. Parsers may inspect filename and optionally 

190 the file at path to refine their confidence level. 

191 

192 A higher score wins. Return None to explicitly decline handling a file 

193 even though the MIME type is listed as supported (e.g. when a feature 

194 flag is disabled, or a required service is not configured). 

195 

196 Parameters 

197 ---------- 

198 mime_type: 

199 The detected MIME type of the file to be parsed. 

200 filename: 

201 The original filename, including extension. 

202 path: 

203 Optional filesystem path to the file. Parsers that need to 

204 inspect file content (e.g. magic-byte sniffing) may use this. 

205 May be None when scoring happens before the file is available locally. 

206 

207 Returns 

208 ------- 

209 int | None 

210 Priority score (higher wins), or None to decline. 

211 """ 

212 ... 

213 

214 # ------------------------------------------------------------------ 

215 # Properties 

216 # ------------------------------------------------------------------ 

217 

218 @property 

219 def can_produce_archive(self) -> bool: 

220 """Whether this parser can produce a searchable PDF archive copy. 

221 

222 If True, the consumption pipeline may request an archive version when 

223 processing the document, subject to the ARCHIVE_FILE_GENERATION 

224 setting. If False, only thumbnail and text extraction are performed. 

225 """ 

226 ... 

227 

228 @property 

229 def requires_pdf_rendition(self) -> bool: 

230 """Whether the parser must produce a PDF for the frontend to display. 

231 

232 True for formats the browser cannot display natively (e.g. DOCX, ODT). 

233 When True, the pipeline always stores the PDF output regardless of the 

234 ARCHIVE_FILE_GENERATION setting, since the original format cannot be 

235 shown to the user. 

236 """ 

237 ... 

238 

239 # ------------------------------------------------------------------ 

240 # Core parsing interface 

241 # ------------------------------------------------------------------ 

242 

243 def configure(self, context: ParserContext) -> None: 

244 """Apply source context before parse(). 

245 

246 Called by the consumer after instantiation and before parse(). 

247 The default implementation is a no-op; parsers override only the 

248 fields they need. 

249 

250 Parameters 

251 ---------- 

252 context: 

253 Immutable context assembled by the consumer for this 

254 specific ingestion event. 

255 """ 

256 ... 

257 

258 def parse( 

259 self, 

260 document_path: Path, 

261 mime_type: str, 

262 *, 

263 produce_archive: bool = True, 

264 ) -> None: 

265 """Parse document_path and populate internal state. 

266 

267 After a successful call, callers retrieve results via get_text, 

268 get_date, and get_archive_path. 

269 

270 Parameters 

271 ---------- 

272 document_path: 

273 Absolute path to the document file to parse. 

274 mime_type: 

275 Detected MIME type of the document. 

276 produce_archive: 

277 When True (the default) and can_produce_archive is also True, 

278 the parser should produce a searchable PDF at the path returned 

279 by get_archive_path. Pass False when only text extraction and 

280 thumbnail generation are required and disk I/O should be minimised. 

281 

282 Raises 

283 ------ 

284 documents.parsers.ParseError 

285 If parsing fails for any reason. 

286 """ 

287 ... 

288 

289 # ------------------------------------------------------------------ 

290 # Result accessors 

291 # ------------------------------------------------------------------ 

292 

293 def get_text(self) -> str: 

294 """Return the plain-text content extracted during parse. 

295 

296 Returns 

297 ------- 

298 str 

299 Extracted text, or an empty string if no text could be found. 

300 """ 

301 ... 

302 

303 def get_date(self) -> datetime.datetime | None: 

304 """Return the document date detected during parse. 

305 

306 Returns 

307 ------- 

308 datetime.datetime | None 

309 Detected document date, or None if no date was found. 

310 """ 

311 ... 

312 

313 def get_archive_path(self) -> Path | None: 

314 """Return the path to the generated archive PDF, or None. 

315 

316 Returns 

317 ------- 

318 Path | None 

319 Path to the searchable PDF archive, or None if no archive was 

320 produced (e.g. because produce_archive=False or the parser does 

321 not support archive generation). 

322 """ 

323 ... 

324 

325 # ------------------------------------------------------------------ 

326 # Thumbnail and metadata 

327 # ------------------------------------------------------------------ 

328 

329 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path: 

330 """Generate and return the path to a thumbnail image for the document. 

331 

332 May be called independently of parse. The returned path must point to 

333 an existing WebP image file inside the parser's temporary working 

334 directory. 

335 

336 Parameters 

337 ---------- 

338 document_path: 

339 Absolute path to the source document. 

340 mime_type: 

341 Detected MIME type of the document. 

342 

343 Returns 

344 ------- 

345 Path 

346 Path to the generated thumbnail image (WebP format preferred). 

347 """ 

348 ... 

349 

350 def get_page_count( 

351 self, 

352 document_path: Path, 

353 mime_type: str, 

354 ) -> int | None: 

355 """Return the number of pages in the document, if determinable. 

356 

357 Parameters 

358 ---------- 

359 document_path: 

360 Absolute path to the source document. 

361 mime_type: 

362 Detected MIME type of the document. 

363 

364 Returns 

365 ------- 

366 int | None 

367 Page count, or None if the parser cannot determine it. 

368 """ 

369 ... 

370 

371 def extract_metadata( 

372 self, 

373 document_path: Path, 

374 mime_type: str, 

375 ) -> list[MetadataEntry]: 

376 """Extract format-specific metadata from the document. 

377 

378 Called by the API view layer on demand — not during the consumption 

379 pipeline. Results are returned to the frontend for per-file display. 

380 

381 For documents with an archive version, this method is called twice: 

382 once for the original file (with its native MIME type) and once for 

383 the archive file (with ``"application/pdf"``). Parsers that produce 

384 archives should handle both cases. 

385 

386 Implementations must not raise. A failure to read metadata is not 

387 fatal — log a warning and return whatever partial results were 

388 collected, or ``[]`` if none. 

389 

390 Parameters 

391 ---------- 

392 document_path: 

393 Absolute path to the file to extract metadata from. 

394 mime_type: 

395 MIME type of the file at ``document_path``. May be 

396 ``"application/pdf"`` when called for the archive version. 

397 

398 Returns 

399 ------- 

400 list[MetadataEntry] 

401 Zero or more metadata entries. Returns ``[]`` if no metadata 

402 could be extracted or the format does not support it. 

403 """ 

404 ... 

405 

406 # ------------------------------------------------------------------ 

407 # Context manager 

408 # ------------------------------------------------------------------ 

409 

410 def __enter__(self) -> Self: 

411 """Enter the parser context, returning the parser instance. 

412 

413 Implementations should perform any resource allocation here if not 

414 done in __init__ (e.g. creating API clients or temp directories). 

415 

416 Returns 

417 ------- 

418 Self 

419 The parser instance itself. 

420 """ 

421 ... 

422 

423 def __exit__( 

424 self, 

425 exc_type: type[BaseException] | None, 

426 exc_val: BaseException | None, 

427 exc_tb: TracebackType | None, 

428 ) -> None: 

429 """Exit the parser context and release all resources. 

430 

431 Implementations must clean up all temporary files and other resources 

432 regardless of whether an exception occurred. 

433 

434 Parameters 

435 ---------- 

436 exc_type: 

437 The exception class, or None if no exception was raised. 

438 exc_val: 

439 The exception instance, or None. 

440 exc_tb: 

441 The traceback, or None. 

442 """ 

443 ...