Coverage for documents/matching.py: 9%

241 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import logging 

4import re 

5from fnmatch import fnmatch 

6from fnmatch import translate as fnmatch_translate 

7from typing import TYPE_CHECKING 

8 

9from rest_framework import serializers 

10 

11from documents.data_models import ConsumableDocument 

12from documents.data_models import DocumentSource 

13from documents.filters import CustomFieldQueryParser 

14from documents.models import Correspondent 

15from documents.models import Document 

16from documents.models import DocumentType 

17from documents.models import MatchingModel 

18from documents.models import StoragePath 

19from documents.models import Tag 

20from documents.models import Workflow 

21from documents.models import WorkflowTrigger 

22from documents.permissions import permitted_object_ids 

23from documents.regex import safe_regex_search 

24 

25if TYPE_CHECKING: 25 ↛ 26line 25 didn't jump to line 26 because the condition on line 25 was never true

26 from django.db.models import QuerySet 

27 

28 from documents.classifier import DocumentClassifier 

29 

30logger = logging.getLogger("paperless.matching") 

31 

32 

33def log_reason( 

34 matching_model: MatchingModel | WorkflowTrigger, 

35 document: Document, 

36 reason: str, 

37) -> None: 

38 class_name = type(matching_model).__name__ 

39 name = ( 

40 matching_model.name if hasattr(matching_model, "name") else str(matching_model) 

41 ) 

42 logger.debug( 

43 f"{class_name} {name} matched on document {document} because {reason}", 

44 ) 

45 

46 

47def match_correspondents(document: Document, classifier: DocumentClassifier, user=None): 

48 pred_id = ( 

49 classifier.predict_correspondent(document.suggestion_content) 

50 if classifier 

51 else None 

52 ) 

53 

54 if user is None and document.owner is not None: 

55 user = document.owner 

56 

57 if user is not None: 

58 correspondents = Correspondent.objects.filter( 

59 id__in=permitted_object_ids(user, Correspondent, "view_correspondent"), 

60 ) 

61 else: 

62 correspondents = Correspondent.objects.all() 

63 

64 return list( 

65 filter( 

66 lambda o: ( 

67 matches(o, document) 

68 or ( 

69 o.pk == pred_id and o.matching_algorithm == MatchingModel.MATCH_AUTO 

70 ) 

71 ), 

72 correspondents, 

73 ), 

74 ) 

75 

76 

77def match_document_types(document: Document, classifier: DocumentClassifier, user=None): 

78 pred_id = ( 

79 classifier.predict_document_type(document.suggestion_content) 

80 if classifier 

81 else None 

82 ) 

83 if user is None and document.owner is not None: 

84 user = document.owner 

85 

86 if user is not None: 

87 document_types = DocumentType.objects.filter( 

88 id__in=permitted_object_ids(user, DocumentType, "view_documenttype"), 

89 ) 

90 else: 

91 document_types = DocumentType.objects.all() 

92 

93 return list( 

94 filter( 

95 lambda o: ( 

96 matches(o, document) 

97 or ( 

98 o.pk == pred_id and o.matching_algorithm == MatchingModel.MATCH_AUTO 

99 ) 

100 ), 

101 document_types, 

102 ), 

103 ) 

104 

105 

106def match_tags(document: Document, classifier: DocumentClassifier, user=None): 

107 predicted_tag_ids = ( 

108 classifier.predict_tags(document.suggestion_content) if classifier else [] 

109 ) 

110 

111 if user is None and document.owner is not None: 

112 user = document.owner 

113 

114 if user is not None: 

115 tags = Tag.objects.filter( 

116 id__in=permitted_object_ids(user, Tag, "view_tag"), 

117 ) 

118 else: 

119 tags = Tag.objects.all() 

120 

121 return list( 

122 filter( 

123 lambda o: ( 

124 matches(o, document) 

125 or ( 

126 o.matching_algorithm == MatchingModel.MATCH_AUTO 

127 and o.pk in predicted_tag_ids 

128 ) 

129 ), 

130 tags, 

131 ), 

132 ) 

133 

134 

135def match_storage_paths(document: Document, classifier: DocumentClassifier, user=None): 

136 pred_id = ( 

137 classifier.predict_storage_path(document.suggestion_content) 

138 if classifier 

139 else None 

140 ) 

141 

142 if user is None and document.owner is not None: 

143 user = document.owner 

144 

145 if user is not None: 

146 storage_paths = StoragePath.objects.filter( 

147 id__in=permitted_object_ids(user, StoragePath, "view_storagepath"), 

148 ) 

149 else: 

150 storage_paths = StoragePath.objects.all() 

151 

152 return list( 

153 filter( 

154 lambda o: ( 

155 matches(o, document) 

156 or ( 

157 o.pk == pred_id and o.matching_algorithm == MatchingModel.MATCH_AUTO 

158 ) 

159 ), 

160 storage_paths, 

161 ), 

162 ) 

163 

164 

165def matches(matching_model: MatchingModel, document: Document): 

166 search_flags = 0 

167 

168 document_content = document.get_effective_content() or "" 

169 

170 # Check that match is not empty 

171 if not matching_model.match.strip(): 

172 return False 

173 

174 if matching_model.is_insensitive: 

175 search_flags = re.IGNORECASE 

176 

177 if matching_model.matching_algorithm == MatchingModel.MATCH_NONE: 

178 return False 

179 

180 elif matching_model.matching_algorithm == MatchingModel.MATCH_ALL: 

181 for word in _split_match(matching_model): 

182 search_result = re.search( 

183 rf"\b{word}\b", 

184 document_content, 

185 flags=search_flags, 

186 ) 

187 if not search_result: 

188 return False 

189 log_reason( 

190 matching_model, 

191 document, 

192 f"it contains all of these words: {matching_model.match}", 

193 ) 

194 return True 

195 

196 elif matching_model.matching_algorithm == MatchingModel.MATCH_ANY: 

197 for word in _split_match(matching_model): 

198 if re.search(rf"\b{word}\b", document_content, flags=search_flags): 

199 log_reason(matching_model, document, f"it contains this word: {word}") 

200 return True 

201 return False 

202 

203 elif matching_model.matching_algorithm == MatchingModel.MATCH_LITERAL: 

204 result = bool( 

205 re.search( 

206 rf"\b{re.escape(matching_model.match)}\b", 

207 document_content, 

208 flags=search_flags, 

209 ), 

210 ) 

211 if result: 

212 log_reason( 

213 matching_model, 

214 document, 

215 f'it contains this string: "{matching_model.match}"', 

216 ) 

217 return result 

218 

219 elif matching_model.matching_algorithm == MatchingModel.MATCH_REGEX: 

220 match = safe_regex_search( 

221 matching_model.match, 

222 document_content, 

223 flags=search_flags, 

224 ) 

225 if match: 

226 log_reason( 

227 matching_model, 

228 document, 

229 f"the string {match.group()} matches the regular expression " 

230 f"{matching_model.match}", 

231 ) 

232 return bool(match) 

233 

234 elif matching_model.matching_algorithm == MatchingModel.MATCH_FUZZY: 

235 from rapidfuzz import fuzz 

236 

237 match = re.sub(r"[^\w\s]", "", matching_model.match) 

238 text = re.sub(r"[^\w\s]", "", document_content) 

239 if matching_model.is_insensitive: 

240 match = match.lower() 

241 text = text.lower() 

242 if fuzz.partial_ratio(match, text, score_cutoff=90): 

243 # TODO: make this better 

244 log_reason( 

245 matching_model, 

246 document, 

247 f"parts of the document content somehow match the string " 

248 f"{matching_model.match}", 

249 ) 

250 return True 

251 else: 

252 return False 

253 

254 elif matching_model.matching_algorithm == MatchingModel.MATCH_AUTO: 

255 # this is done elsewhere. 

256 return False 

257 

258 else: 

259 raise NotImplementedError("Unsupported matching algorithm") 

260 

261 

262def _split_match(matching_model): 

263 """ 

264 Splits the match to individual keywords, getting rid of unnecessary 

265 spaces and grouping quoted words together. 

266 

267 Example: 

268 ' some random words "with quotes " and spaces' 

269 ==> 

270 ["some", "random", "words", "with+quotes", "and", "spaces"] 

271 """ 

272 findterms = re.compile(r'"([^"]+)"|(\S+)').findall 

273 normspace = re.compile(r"\s+").sub 

274 return [ 

275 # normspace(" ", (t[0] or t[1]).strip()).replace(" ", r"\s+") 

276 re.escape(normspace(" ", (t[0] or t[1]).strip())).replace(r"\ ", r"\s+") 

277 for t in findterms(matching_model.match) 

278 ] 

279 

280 

281def consumable_document_matches_workflow( 

282 document: ConsumableDocument, 

283 trigger: WorkflowTrigger, 

284) -> tuple[bool, str]: 

285 """ 

286 Returns True if the ConsumableDocument matches all filters from the workflow trigger, 

287 False otherwise. Includes a reason if doesn't match 

288 """ 

289 

290 trigger_matched = True 

291 reason = "" 

292 

293 # Document source vs trigger source 

294 if len(trigger.sources) > 0 and document.source not in [ 

295 int(x) for x in list(trigger.sources) 

296 ]: 

297 reason = ( 

298 f"Document source {document.source.name} not in" 

299 f" {[DocumentSource(int(x)).name for x in trigger.sources]}" 

300 ) 

301 trigger_matched = False 

302 

303 # Document mail rule vs trigger mail rule 

304 if ( 

305 trigger.filter_mailrule is not None 

306 and document.mailrule_id != trigger.filter_mailrule.pk 

307 ): 

308 reason = ( 

309 f"Document mail rule {document.mailrule_id} != {trigger.filter_mailrule.pk}" 

310 ) 

311 trigger_matched = False 

312 

313 # Document filename vs trigger filename 

314 if ( 

315 trigger.filter_filename is not None 

316 and len(trigger.filter_filename) > 0 

317 and not fnmatch( 

318 document.original_file.name.lower(), 

319 trigger.filter_filename.lower(), 

320 ) 

321 ): 

322 reason = ( 

323 f"Document filename {document.original_file.name} does not match" 

324 f" {trigger.filter_filename.lower()}" 

325 ) 

326 trigger_matched = False 

327 

328 # Document path vs trigger path 

329 

330 # Use the original_path if set, else us the original_file 

331 match_against = ( 

332 document.original_path 

333 if document.original_path is not None 

334 else document.original_file 

335 ) 

336 

337 if ( 

338 trigger.filter_path is not None 

339 and len(trigger.filter_path) > 0 

340 and not fnmatch( 

341 match_against, 

342 trigger.filter_path, 

343 ) 

344 ): 

345 reason = ( 

346 f"Document path {document.original_file}" 

347 f" does not match {trigger.filter_path}" 

348 ) 

349 trigger_matched = False 

350 

351 return (trigger_matched, reason) 

352 

353 

354def existing_document_matches_workflow( 

355 document: Document, 

356 trigger: WorkflowTrigger, 

357) -> tuple[bool, str | None]: 

358 """ 

359 Returns True if the Document matches all filters from the workflow trigger, 

360 False otherwise. Includes a reason if doesn't match 

361 """ 

362 

363 # Check content matching algorithm 

364 if trigger.matching_algorithm > MatchingModel.MATCH_NONE and not matches( 

365 trigger, 

366 document, 

367 ): 

368 return ( 

369 False, 

370 f"Document content matching settings for algorithm '{trigger.matching_algorithm}' did not match", 

371 ) 

372 

373 # Check if any tag filters exist to determine if we need to load document tags 

374 trigger_has_tags_qs = trigger.filter_has_tags.all() 

375 trigger_has_all_tags_qs = trigger.filter_has_all_tags.all() 

376 trigger_has_not_tags_qs = trigger.filter_has_not_tags.all() 

377 

378 has_tags_filter = trigger_has_tags_qs.exists() 

379 has_all_tags_filter = trigger_has_all_tags_qs.exists() 

380 has_not_tags_filter = trigger_has_not_tags_qs.exists() 

381 

382 # Load document tags once if any tag filters exist 

383 document_tag_ids = None 

384 if has_tags_filter or has_all_tags_filter or has_not_tags_filter: 

385 document_tag_ids = set(document.tags.values_list("id", flat=True)) 

386 

387 # Document tags vs trigger has_tags (any of) 

388 if has_tags_filter: 

389 trigger_has_tag_ids = set(trigger_has_tags_qs.values_list("id", flat=True)) 

390 if not (document_tag_ids & trigger_has_tag_ids): 

391 # For error message, load the actual tag objects 

392 return ( 

393 False, 

394 f"Document tags {list(document.tags.all())} do not include {list(trigger_has_tags_qs)}", 

395 ) 

396 

397 # Document tags vs trigger has_all_tags (all of) 

398 if has_all_tags_filter: 

399 required_tag_ids = set(trigger_has_all_tags_qs.values_list("id", flat=True)) 

400 if not required_tag_ids.issubset(document_tag_ids): 

401 return ( 

402 False, 

403 f"Document tags {list(document.tags.all())} do not contain all of {list(trigger_has_all_tags_qs)}", 

404 ) 

405 

406 # Document tags vs trigger has_not_tags (none of) 

407 if has_not_tags_filter: 

408 excluded_tag_ids = set(trigger_has_not_tags_qs.values_list("id", flat=True)) 

409 if document_tag_ids & excluded_tag_ids: 

410 return ( 

411 False, 

412 f"Document tags {list(document.tags.all())} include excluded tags {list(trigger_has_not_tags_qs)}", 

413 ) 

414 

415 allowed_correspondent_ids = set( 

416 trigger.filter_has_any_correspondents.values_list("id", flat=True), 

417 ) 

418 if ( 

419 allowed_correspondent_ids 

420 and document.correspondent_id not in allowed_correspondent_ids 

421 ): 

422 return ( 

423 False, 

424 f"Document correspondent {document.correspondent} is not one of {list(trigger.filter_has_any_correspondents.all())}", 

425 ) 

426 

427 # Document correspondent vs trigger has_correspondent 

428 if ( 

429 trigger.filter_has_correspondent_id is not None 

430 and document.correspondent_id != trigger.filter_has_correspondent_id 

431 ): 

432 return ( 

433 False, 

434 f"Document correspondent {document.correspondent} does not match {trigger.filter_has_correspondent}", 

435 ) 

436 

437 if ( 

438 document.correspondent_id 

439 and trigger.filter_has_not_correspondents.filter( 

440 id=document.correspondent_id, 

441 ).exists() 

442 ): 

443 return ( 

444 False, 

445 f"Document correspondent {document.correspondent} is excluded by {list(trigger.filter_has_not_correspondents.all())}", 

446 ) 

447 

448 allowed_document_type_ids = set( 

449 trigger.filter_has_any_document_types.values_list("id", flat=True), 

450 ) 

451 if allowed_document_type_ids and ( 

452 document.document_type_id not in allowed_document_type_ids 

453 ): 

454 return ( 

455 False, 

456 f"Document doc type {document.document_type} is not one of {list(trigger.filter_has_any_document_types.all())}", 

457 ) 

458 

459 # Document document_type vs trigger has_document_type 

460 if ( 

461 trigger.filter_has_document_type_id is not None 

462 and document.document_type_id != trigger.filter_has_document_type_id 

463 ): 

464 return ( 

465 False, 

466 f"Document doc type {document.document_type} does not match {trigger.filter_has_document_type}", 

467 ) 

468 

469 if ( 

470 document.document_type_id 

471 and trigger.filter_has_not_document_types.filter( 

472 id=document.document_type_id, 

473 ).exists() 

474 ): 

475 return ( 

476 False, 

477 f"Document doc type {document.document_type} is excluded by {list(trigger.filter_has_not_document_types.all())}", 

478 ) 

479 

480 allowed_storage_path_ids = set( 

481 trigger.filter_has_any_storage_paths.values_list("id", flat=True), 

482 ) 

483 if allowed_storage_path_ids and ( 

484 document.storage_path_id not in allowed_storage_path_ids 

485 ): 

486 return ( 

487 False, 

488 f"Document storage path {document.storage_path} is not one of {list(trigger.filter_has_any_storage_paths.all())}", 

489 ) 

490 

491 # Document storage_path vs trigger has_storage_path 

492 if ( 

493 trigger.filter_has_storage_path_id is not None 

494 and document.storage_path_id != trigger.filter_has_storage_path_id 

495 ): 

496 return ( 

497 False, 

498 f"Document storage path {document.storage_path} does not match {trigger.filter_has_storage_path}", 

499 ) 

500 

501 if ( 

502 document.storage_path_id 

503 and trigger.filter_has_not_storage_paths.filter( 

504 id=document.storage_path_id, 

505 ).exists() 

506 ): 

507 return ( 

508 False, 

509 f"Document storage path {document.storage_path} is excluded by {list(trigger.filter_has_not_storage_paths.all())}", 

510 ) 

511 

512 # Custom field query check 

513 if trigger.filter_custom_field_query: 

514 parser = CustomFieldQueryParser("filter_custom_field_query") 

515 try: 

516 custom_field_q, annotations = parser.parse( 

517 trigger.filter_custom_field_query, 

518 ) 

519 except serializers.ValidationError: 

520 return (False, "Invalid custom field query configuration") 

521 

522 qs = ( 

523 Document.objects.filter(id=document.id) 

524 .annotate(**annotations) 

525 .filter(custom_field_q) 

526 ) 

527 if not qs.exists(): 

528 return ( 

529 False, 

530 "Document custom fields do not match the configured custom field query", 

531 ) 

532 

533 # Document original_filename vs trigger filename 

534 if ( 

535 trigger.filter_filename is not None 

536 and len(trigger.filter_filename) > 0 

537 and document.original_filename is not None 

538 and not fnmatch( 

539 document.original_filename.lower(), 

540 trigger.filter_filename.lower(), 

541 ) 

542 ): 

543 return ( 

544 False, 

545 f"Document filename {document.original_filename} does not match {trigger.filter_filename.lower()}", 

546 ) 

547 

548 return (True, None) 

549 

550 

551def prefilter_documents_by_workflowtrigger( 

552 documents: QuerySet[Document], 

553 trigger: WorkflowTrigger, 

554) -> QuerySet[Document]: 

555 """ 

556 To prevent scheduled workflows checking every document, we prefilter the 

557 documents by the workflow trigger filters. This is done before e.g. 

558 document_matches_workflow in run_workflows 

559 """ 

560 

561 # Filter for documents that have AT LEAST ONE of the specified tags. 

562 if trigger.filter_has_tags.exists(): 

563 documents = documents.filter(tags__in=trigger.filter_has_tags.all()).distinct() 

564 

565 # Filter for documents that have ALL of the specified tags. 

566 if trigger.filter_has_all_tags.exists(): 

567 for tag in trigger.filter_has_all_tags.all(): 

568 documents = documents.filter(tags=tag) 

569 # Multiple JOINs can create duplicate results. 

570 documents = documents.distinct() 

571 

572 # Exclude documents that have ANY of the specified tags. 

573 if trigger.filter_has_not_tags.exists(): 

574 documents = documents.exclude(tags__in=trigger.filter_has_not_tags.all()) 

575 

576 # Correspondent, DocumentType, etc. filtering 

577 

578 if trigger.filter_has_any_correspondents.exists(): 

579 documents = documents.filter( 

580 correspondent__in=trigger.filter_has_any_correspondents.all(), 

581 ) 

582 if trigger.filter_has_correspondent is not None: 

583 documents = documents.filter( 

584 correspondent=trigger.filter_has_correspondent, 

585 ) 

586 if trigger.filter_has_not_correspondents.exists(): 

587 documents = documents.exclude( 

588 correspondent__in=trigger.filter_has_not_correspondents.all(), 

589 ) 

590 

591 if trigger.filter_has_any_document_types.exists(): 

592 documents = documents.filter( 

593 document_type__in=trigger.filter_has_any_document_types.all(), 

594 ) 

595 if trigger.filter_has_document_type is not None: 

596 documents = documents.filter( 

597 document_type=trigger.filter_has_document_type, 

598 ) 

599 if trigger.filter_has_not_document_types.exists(): 

600 documents = documents.exclude( 

601 document_type__in=trigger.filter_has_not_document_types.all(), 

602 ) 

603 

604 if trigger.filter_has_any_storage_paths.exists(): 

605 documents = documents.filter( 

606 storage_path__in=trigger.filter_has_any_storage_paths.all(), 

607 ) 

608 if trigger.filter_has_storage_path is not None: 

609 documents = documents.filter( 

610 storage_path=trigger.filter_has_storage_path, 

611 ) 

612 if trigger.filter_has_not_storage_paths.exists(): 

613 documents = documents.exclude( 

614 storage_path__in=trigger.filter_has_not_storage_paths.all(), 

615 ) 

616 

617 # Custom Field & Filename Filtering 

618 

619 if trigger.filter_custom_field_query: 

620 parser = CustomFieldQueryParser("filter_custom_field_query") 

621 try: 

622 custom_field_q, annotations = parser.parse( 

623 trigger.filter_custom_field_query, 

624 ) 

625 except serializers.ValidationError: 

626 return documents.none() 

627 

628 documents = documents.annotate(**annotations).filter(custom_field_q) 

629 

630 if trigger.filter_filename: 

631 regex = fnmatch_translate(trigger.filter_filename).lstrip("^").rstrip("$") 

632 documents = documents.filter(original_filename__iregex=regex) 

633 

634 return documents 

635 

636 

637def document_matches_workflow( 

638 document: ConsumableDocument | Document, 

639 workflow: Workflow, 

640 trigger_type: WorkflowTrigger.WorkflowTriggerType, 

641) -> bool: 

642 """ 

643 Returns True if the ConsumableDocument or Document matches all filters and 

644 settings from the workflow trigger, False otherwise 

645 """ 

646 

647 triggers_queryset = ( 

648 workflow.triggers.filter( 

649 type=trigger_type, 

650 ) 

651 .select_related( 

652 "filter_mailrule", 

653 "filter_has_document_type", 

654 "filter_has_correspondent", 

655 "filter_has_storage_path", 

656 "schedule_date_custom_field", 

657 ) 

658 .prefetch_related( 

659 "filter_has_tags", 

660 "filter_has_all_tags", 

661 "filter_has_not_tags", 

662 "filter_has_any_document_types", 

663 "filter_has_not_document_types", 

664 "filter_has_any_correspondents", 

665 "filter_has_not_correspondents", 

666 "filter_has_any_storage_paths", 

667 "filter_has_not_storage_paths", 

668 ) 

669 ) 

670 

671 trigger_matched = True 

672 if not triggers_queryset.exists(): 

673 trigger_matched = False 

674 logger.info(f"Document did not match {workflow}") 

675 logger.debug(f"No matching triggers with type {trigger_type} found") 

676 else: 

677 for trigger in triggers_queryset: 

678 if trigger_type == WorkflowTrigger.WorkflowTriggerType.CONSUMPTION: 

679 trigger_matched, reason = consumable_document_matches_workflow( 

680 document, 

681 trigger, 

682 ) 

683 elif ( 

684 trigger_type == WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED 

685 or trigger_type == WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED 

686 or trigger_type == WorkflowTrigger.WorkflowTriggerType.SCHEDULED 

687 ): 

688 trigger_matched, reason = existing_document_matches_workflow( 

689 document, 

690 trigger, 

691 ) 

692 else: 

693 # New trigger types need to be explicitly checked above 

694 raise Exception(f"Trigger type {trigger_type} not yet supported") 

695 

696 if trigger_matched: 

697 logger.info(f"Document matched {trigger} from {workflow}") 

698 # matched, bail early 

699 return True 

700 else: 

701 logger.info(f"Document did not match {workflow}") 

702 logger.debug(reason) 

703 

704 return trigger_matched