Coverage for documents/matching.py: 9%
241 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import logging
4import re
5from fnmatch import fnmatch
6from fnmatch import translate as fnmatch_translate
7from typing import TYPE_CHECKING
9from rest_framework import serializers
11from documents.data_models import ConsumableDocument
12from documents.data_models import DocumentSource
13from documents.filters import CustomFieldQueryParser
14from documents.models import Correspondent
15from documents.models import Document
16from documents.models import DocumentType
17from documents.models import MatchingModel
18from documents.models import StoragePath
19from documents.models import Tag
20from documents.models import Workflow
21from documents.models import WorkflowTrigger
22from documents.permissions import permitted_object_ids
23from documents.regex import safe_regex_search
25if TYPE_CHECKING: 25 ↛ 26line 25 didn't jump to line 26 because the condition on line 25 was never true
26 from django.db.models import QuerySet
28 from documents.classifier import DocumentClassifier
30logger = logging.getLogger("paperless.matching")
33def log_reason(
34 matching_model: MatchingModel | WorkflowTrigger,
35 document: Document,
36 reason: str,
37) -> None:
38 class_name = type(matching_model).__name__
39 name = (
40 matching_model.name if hasattr(matching_model, "name") else str(matching_model)
41 )
42 logger.debug(
43 f"{class_name} {name} matched on document {document} because {reason}",
44 )
47def match_correspondents(document: Document, classifier: DocumentClassifier, user=None):
48 pred_id = (
49 classifier.predict_correspondent(document.suggestion_content)
50 if classifier
51 else None
52 )
54 if user is None and document.owner is not None:
55 user = document.owner
57 if user is not None:
58 correspondents = Correspondent.objects.filter(
59 id__in=permitted_object_ids(user, Correspondent, "view_correspondent"),
60 )
61 else:
62 correspondents = Correspondent.objects.all()
64 return list(
65 filter(
66 lambda o: (
67 matches(o, document)
68 or (
69 o.pk == pred_id and o.matching_algorithm == MatchingModel.MATCH_AUTO
70 )
71 ),
72 correspondents,
73 ),
74 )
77def match_document_types(document: Document, classifier: DocumentClassifier, user=None):
78 pred_id = (
79 classifier.predict_document_type(document.suggestion_content)
80 if classifier
81 else None
82 )
83 if user is None and document.owner is not None:
84 user = document.owner
86 if user is not None:
87 document_types = DocumentType.objects.filter(
88 id__in=permitted_object_ids(user, DocumentType, "view_documenttype"),
89 )
90 else:
91 document_types = DocumentType.objects.all()
93 return list(
94 filter(
95 lambda o: (
96 matches(o, document)
97 or (
98 o.pk == pred_id and o.matching_algorithm == MatchingModel.MATCH_AUTO
99 )
100 ),
101 document_types,
102 ),
103 )
106def match_tags(document: Document, classifier: DocumentClassifier, user=None):
107 predicted_tag_ids = (
108 classifier.predict_tags(document.suggestion_content) if classifier else []
109 )
111 if user is None and document.owner is not None:
112 user = document.owner
114 if user is not None:
115 tags = Tag.objects.filter(
116 id__in=permitted_object_ids(user, Tag, "view_tag"),
117 )
118 else:
119 tags = Tag.objects.all()
121 return list(
122 filter(
123 lambda o: (
124 matches(o, document)
125 or (
126 o.matching_algorithm == MatchingModel.MATCH_AUTO
127 and o.pk in predicted_tag_ids
128 )
129 ),
130 tags,
131 ),
132 )
135def match_storage_paths(document: Document, classifier: DocumentClassifier, user=None):
136 pred_id = (
137 classifier.predict_storage_path(document.suggestion_content)
138 if classifier
139 else None
140 )
142 if user is None and document.owner is not None:
143 user = document.owner
145 if user is not None:
146 storage_paths = StoragePath.objects.filter(
147 id__in=permitted_object_ids(user, StoragePath, "view_storagepath"),
148 )
149 else:
150 storage_paths = StoragePath.objects.all()
152 return list(
153 filter(
154 lambda o: (
155 matches(o, document)
156 or (
157 o.pk == pred_id and o.matching_algorithm == MatchingModel.MATCH_AUTO
158 )
159 ),
160 storage_paths,
161 ),
162 )
165def matches(matching_model: MatchingModel, document: Document):
166 search_flags = 0
168 document_content = document.get_effective_content() or ""
170 # Check that match is not empty
171 if not matching_model.match.strip():
172 return False
174 if matching_model.is_insensitive:
175 search_flags = re.IGNORECASE
177 if matching_model.matching_algorithm == MatchingModel.MATCH_NONE:
178 return False
180 elif matching_model.matching_algorithm == MatchingModel.MATCH_ALL:
181 for word in _split_match(matching_model):
182 search_result = re.search(
183 rf"\b{word}\b",
184 document_content,
185 flags=search_flags,
186 )
187 if not search_result:
188 return False
189 log_reason(
190 matching_model,
191 document,
192 f"it contains all of these words: {matching_model.match}",
193 )
194 return True
196 elif matching_model.matching_algorithm == MatchingModel.MATCH_ANY:
197 for word in _split_match(matching_model):
198 if re.search(rf"\b{word}\b", document_content, flags=search_flags):
199 log_reason(matching_model, document, f"it contains this word: {word}")
200 return True
201 return False
203 elif matching_model.matching_algorithm == MatchingModel.MATCH_LITERAL:
204 result = bool(
205 re.search(
206 rf"\b{re.escape(matching_model.match)}\b",
207 document_content,
208 flags=search_flags,
209 ),
210 )
211 if result:
212 log_reason(
213 matching_model,
214 document,
215 f'it contains this string: "{matching_model.match}"',
216 )
217 return result
219 elif matching_model.matching_algorithm == MatchingModel.MATCH_REGEX:
220 match = safe_regex_search(
221 matching_model.match,
222 document_content,
223 flags=search_flags,
224 )
225 if match:
226 log_reason(
227 matching_model,
228 document,
229 f"the string {match.group()} matches the regular expression "
230 f"{matching_model.match}",
231 )
232 return bool(match)
234 elif matching_model.matching_algorithm == MatchingModel.MATCH_FUZZY:
235 from rapidfuzz import fuzz
237 match = re.sub(r"[^\w\s]", "", matching_model.match)
238 text = re.sub(r"[^\w\s]", "", document_content)
239 if matching_model.is_insensitive:
240 match = match.lower()
241 text = text.lower()
242 if fuzz.partial_ratio(match, text, score_cutoff=90):
243 # TODO: make this better
244 log_reason(
245 matching_model,
246 document,
247 f"parts of the document content somehow match the string "
248 f"{matching_model.match}",
249 )
250 return True
251 else:
252 return False
254 elif matching_model.matching_algorithm == MatchingModel.MATCH_AUTO:
255 # this is done elsewhere.
256 return False
258 else:
259 raise NotImplementedError("Unsupported matching algorithm")
262def _split_match(matching_model):
263 """
264 Splits the match to individual keywords, getting rid of unnecessary
265 spaces and grouping quoted words together.
267 Example:
268 ' some random words "with quotes " and spaces'
269 ==>
270 ["some", "random", "words", "with+quotes", "and", "spaces"]
271 """
272 findterms = re.compile(r'"([^"]+)"|(\S+)').findall
273 normspace = re.compile(r"\s+").sub
274 return [
275 # normspace(" ", (t[0] or t[1]).strip()).replace(" ", r"\s+")
276 re.escape(normspace(" ", (t[0] or t[1]).strip())).replace(r"\ ", r"\s+")
277 for t in findterms(matching_model.match)
278 ]
281def consumable_document_matches_workflow(
282 document: ConsumableDocument,
283 trigger: WorkflowTrigger,
284) -> tuple[bool, str]:
285 """
286 Returns True if the ConsumableDocument matches all filters from the workflow trigger,
287 False otherwise. Includes a reason if doesn't match
288 """
290 trigger_matched = True
291 reason = ""
293 # Document source vs trigger source
294 if len(trigger.sources) > 0 and document.source not in [
295 int(x) for x in list(trigger.sources)
296 ]:
297 reason = (
298 f"Document source {document.source.name} not in"
299 f" {[DocumentSource(int(x)).name for x in trigger.sources]}"
300 )
301 trigger_matched = False
303 # Document mail rule vs trigger mail rule
304 if (
305 trigger.filter_mailrule is not None
306 and document.mailrule_id != trigger.filter_mailrule.pk
307 ):
308 reason = (
309 f"Document mail rule {document.mailrule_id} != {trigger.filter_mailrule.pk}"
310 )
311 trigger_matched = False
313 # Document filename vs trigger filename
314 if (
315 trigger.filter_filename is not None
316 and len(trigger.filter_filename) > 0
317 and not fnmatch(
318 document.original_file.name.lower(),
319 trigger.filter_filename.lower(),
320 )
321 ):
322 reason = (
323 f"Document filename {document.original_file.name} does not match"
324 f" {trigger.filter_filename.lower()}"
325 )
326 trigger_matched = False
328 # Document path vs trigger path
330 # Use the original_path if set, else us the original_file
331 match_against = (
332 document.original_path
333 if document.original_path is not None
334 else document.original_file
335 )
337 if (
338 trigger.filter_path is not None
339 and len(trigger.filter_path) > 0
340 and not fnmatch(
341 match_against,
342 trigger.filter_path,
343 )
344 ):
345 reason = (
346 f"Document path {document.original_file}"
347 f" does not match {trigger.filter_path}"
348 )
349 trigger_matched = False
351 return (trigger_matched, reason)
354def existing_document_matches_workflow(
355 document: Document,
356 trigger: WorkflowTrigger,
357) -> tuple[bool, str | None]:
358 """
359 Returns True if the Document matches all filters from the workflow trigger,
360 False otherwise. Includes a reason if doesn't match
361 """
363 # Check content matching algorithm
364 if trigger.matching_algorithm > MatchingModel.MATCH_NONE and not matches(
365 trigger,
366 document,
367 ):
368 return (
369 False,
370 f"Document content matching settings for algorithm '{trigger.matching_algorithm}' did not match",
371 )
373 # Check if any tag filters exist to determine if we need to load document tags
374 trigger_has_tags_qs = trigger.filter_has_tags.all()
375 trigger_has_all_tags_qs = trigger.filter_has_all_tags.all()
376 trigger_has_not_tags_qs = trigger.filter_has_not_tags.all()
378 has_tags_filter = trigger_has_tags_qs.exists()
379 has_all_tags_filter = trigger_has_all_tags_qs.exists()
380 has_not_tags_filter = trigger_has_not_tags_qs.exists()
382 # Load document tags once if any tag filters exist
383 document_tag_ids = None
384 if has_tags_filter or has_all_tags_filter or has_not_tags_filter:
385 document_tag_ids = set(document.tags.values_list("id", flat=True))
387 # Document tags vs trigger has_tags (any of)
388 if has_tags_filter:
389 trigger_has_tag_ids = set(trigger_has_tags_qs.values_list("id", flat=True))
390 if not (document_tag_ids & trigger_has_tag_ids):
391 # For error message, load the actual tag objects
392 return (
393 False,
394 f"Document tags {list(document.tags.all())} do not include {list(trigger_has_tags_qs)}",
395 )
397 # Document tags vs trigger has_all_tags (all of)
398 if has_all_tags_filter:
399 required_tag_ids = set(trigger_has_all_tags_qs.values_list("id", flat=True))
400 if not required_tag_ids.issubset(document_tag_ids):
401 return (
402 False,
403 f"Document tags {list(document.tags.all())} do not contain all of {list(trigger_has_all_tags_qs)}",
404 )
406 # Document tags vs trigger has_not_tags (none of)
407 if has_not_tags_filter:
408 excluded_tag_ids = set(trigger_has_not_tags_qs.values_list("id", flat=True))
409 if document_tag_ids & excluded_tag_ids:
410 return (
411 False,
412 f"Document tags {list(document.tags.all())} include excluded tags {list(trigger_has_not_tags_qs)}",
413 )
415 allowed_correspondent_ids = set(
416 trigger.filter_has_any_correspondents.values_list("id", flat=True),
417 )
418 if (
419 allowed_correspondent_ids
420 and document.correspondent_id not in allowed_correspondent_ids
421 ):
422 return (
423 False,
424 f"Document correspondent {document.correspondent} is not one of {list(trigger.filter_has_any_correspondents.all())}",
425 )
427 # Document correspondent vs trigger has_correspondent
428 if (
429 trigger.filter_has_correspondent_id is not None
430 and document.correspondent_id != trigger.filter_has_correspondent_id
431 ):
432 return (
433 False,
434 f"Document correspondent {document.correspondent} does not match {trigger.filter_has_correspondent}",
435 )
437 if (
438 document.correspondent_id
439 and trigger.filter_has_not_correspondents.filter(
440 id=document.correspondent_id,
441 ).exists()
442 ):
443 return (
444 False,
445 f"Document correspondent {document.correspondent} is excluded by {list(trigger.filter_has_not_correspondents.all())}",
446 )
448 allowed_document_type_ids = set(
449 trigger.filter_has_any_document_types.values_list("id", flat=True),
450 )
451 if allowed_document_type_ids and (
452 document.document_type_id not in allowed_document_type_ids
453 ):
454 return (
455 False,
456 f"Document doc type {document.document_type} is not one of {list(trigger.filter_has_any_document_types.all())}",
457 )
459 # Document document_type vs trigger has_document_type
460 if (
461 trigger.filter_has_document_type_id is not None
462 and document.document_type_id != trigger.filter_has_document_type_id
463 ):
464 return (
465 False,
466 f"Document doc type {document.document_type} does not match {trigger.filter_has_document_type}",
467 )
469 if (
470 document.document_type_id
471 and trigger.filter_has_not_document_types.filter(
472 id=document.document_type_id,
473 ).exists()
474 ):
475 return (
476 False,
477 f"Document doc type {document.document_type} is excluded by {list(trigger.filter_has_not_document_types.all())}",
478 )
480 allowed_storage_path_ids = set(
481 trigger.filter_has_any_storage_paths.values_list("id", flat=True),
482 )
483 if allowed_storage_path_ids and (
484 document.storage_path_id not in allowed_storage_path_ids
485 ):
486 return (
487 False,
488 f"Document storage path {document.storage_path} is not one of {list(trigger.filter_has_any_storage_paths.all())}",
489 )
491 # Document storage_path vs trigger has_storage_path
492 if (
493 trigger.filter_has_storage_path_id is not None
494 and document.storage_path_id != trigger.filter_has_storage_path_id
495 ):
496 return (
497 False,
498 f"Document storage path {document.storage_path} does not match {trigger.filter_has_storage_path}",
499 )
501 if (
502 document.storage_path_id
503 and trigger.filter_has_not_storage_paths.filter(
504 id=document.storage_path_id,
505 ).exists()
506 ):
507 return (
508 False,
509 f"Document storage path {document.storage_path} is excluded by {list(trigger.filter_has_not_storage_paths.all())}",
510 )
512 # Custom field query check
513 if trigger.filter_custom_field_query:
514 parser = CustomFieldQueryParser("filter_custom_field_query")
515 try:
516 custom_field_q, annotations = parser.parse(
517 trigger.filter_custom_field_query,
518 )
519 except serializers.ValidationError:
520 return (False, "Invalid custom field query configuration")
522 qs = (
523 Document.objects.filter(id=document.id)
524 .annotate(**annotations)
525 .filter(custom_field_q)
526 )
527 if not qs.exists():
528 return (
529 False,
530 "Document custom fields do not match the configured custom field query",
531 )
533 # Document original_filename vs trigger filename
534 if (
535 trigger.filter_filename is not None
536 and len(trigger.filter_filename) > 0
537 and document.original_filename is not None
538 and not fnmatch(
539 document.original_filename.lower(),
540 trigger.filter_filename.lower(),
541 )
542 ):
543 return (
544 False,
545 f"Document filename {document.original_filename} does not match {trigger.filter_filename.lower()}",
546 )
548 return (True, None)
551def prefilter_documents_by_workflowtrigger(
552 documents: QuerySet[Document],
553 trigger: WorkflowTrigger,
554) -> QuerySet[Document]:
555 """
556 To prevent scheduled workflows checking every document, we prefilter the
557 documents by the workflow trigger filters. This is done before e.g.
558 document_matches_workflow in run_workflows
559 """
561 # Filter for documents that have AT LEAST ONE of the specified tags.
562 if trigger.filter_has_tags.exists():
563 documents = documents.filter(tags__in=trigger.filter_has_tags.all()).distinct()
565 # Filter for documents that have ALL of the specified tags.
566 if trigger.filter_has_all_tags.exists():
567 for tag in trigger.filter_has_all_tags.all():
568 documents = documents.filter(tags=tag)
569 # Multiple JOINs can create duplicate results.
570 documents = documents.distinct()
572 # Exclude documents that have ANY of the specified tags.
573 if trigger.filter_has_not_tags.exists():
574 documents = documents.exclude(tags__in=trigger.filter_has_not_tags.all())
576 # Correspondent, DocumentType, etc. filtering
578 if trigger.filter_has_any_correspondents.exists():
579 documents = documents.filter(
580 correspondent__in=trigger.filter_has_any_correspondents.all(),
581 )
582 if trigger.filter_has_correspondent is not None:
583 documents = documents.filter(
584 correspondent=trigger.filter_has_correspondent,
585 )
586 if trigger.filter_has_not_correspondents.exists():
587 documents = documents.exclude(
588 correspondent__in=trigger.filter_has_not_correspondents.all(),
589 )
591 if trigger.filter_has_any_document_types.exists():
592 documents = documents.filter(
593 document_type__in=trigger.filter_has_any_document_types.all(),
594 )
595 if trigger.filter_has_document_type is not None:
596 documents = documents.filter(
597 document_type=trigger.filter_has_document_type,
598 )
599 if trigger.filter_has_not_document_types.exists():
600 documents = documents.exclude(
601 document_type__in=trigger.filter_has_not_document_types.all(),
602 )
604 if trigger.filter_has_any_storage_paths.exists():
605 documents = documents.filter(
606 storage_path__in=trigger.filter_has_any_storage_paths.all(),
607 )
608 if trigger.filter_has_storage_path is not None:
609 documents = documents.filter(
610 storage_path=trigger.filter_has_storage_path,
611 )
612 if trigger.filter_has_not_storage_paths.exists():
613 documents = documents.exclude(
614 storage_path__in=trigger.filter_has_not_storage_paths.all(),
615 )
617 # Custom Field & Filename Filtering
619 if trigger.filter_custom_field_query:
620 parser = CustomFieldQueryParser("filter_custom_field_query")
621 try:
622 custom_field_q, annotations = parser.parse(
623 trigger.filter_custom_field_query,
624 )
625 except serializers.ValidationError:
626 return documents.none()
628 documents = documents.annotate(**annotations).filter(custom_field_q)
630 if trigger.filter_filename:
631 regex = fnmatch_translate(trigger.filter_filename).lstrip("^").rstrip("$")
632 documents = documents.filter(original_filename__iregex=regex)
634 return documents
637def document_matches_workflow(
638 document: ConsumableDocument | Document,
639 workflow: Workflow,
640 trigger_type: WorkflowTrigger.WorkflowTriggerType,
641) -> bool:
642 """
643 Returns True if the ConsumableDocument or Document matches all filters and
644 settings from the workflow trigger, False otherwise
645 """
647 triggers_queryset = (
648 workflow.triggers.filter(
649 type=trigger_type,
650 )
651 .select_related(
652 "filter_mailrule",
653 "filter_has_document_type",
654 "filter_has_correspondent",
655 "filter_has_storage_path",
656 "schedule_date_custom_field",
657 )
658 .prefetch_related(
659 "filter_has_tags",
660 "filter_has_all_tags",
661 "filter_has_not_tags",
662 "filter_has_any_document_types",
663 "filter_has_not_document_types",
664 "filter_has_any_correspondents",
665 "filter_has_not_correspondents",
666 "filter_has_any_storage_paths",
667 "filter_has_not_storage_paths",
668 )
669 )
671 trigger_matched = True
672 if not triggers_queryset.exists():
673 trigger_matched = False
674 logger.info(f"Document did not match {workflow}")
675 logger.debug(f"No matching triggers with type {trigger_type} found")
676 else:
677 for trigger in triggers_queryset:
678 if trigger_type == WorkflowTrigger.WorkflowTriggerType.CONSUMPTION:
679 trigger_matched, reason = consumable_document_matches_workflow(
680 document,
681 trigger,
682 )
683 elif (
684 trigger_type == WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED
685 or trigger_type == WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED
686 or trigger_type == WorkflowTrigger.WorkflowTriggerType.SCHEDULED
687 ):
688 trigger_matched, reason = existing_document_matches_workflow(
689 document,
690 trigger,
691 )
692 else:
693 # New trigger types need to be explicitly checked above
694 raise Exception(f"Trigger type {trigger_type} not yet supported")
696 if trigger_matched:
697 logger.info(f"Document matched {trigger} from {workflow}")
698 # matched, bail early
699 return True
700 else:
701 logger.info(f"Document did not match {workflow}")
702 logger.debug(reason)
704 return trigger_matched