Coverage for documents/management/commands/document_fuzzy_match.py: 0%
107 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1import dataclasses
2from itertools import combinations
3from typing import Final
5import rapidfuzz
6from django.core.management import CommandError
7from rich.panel import Panel
8from rich.table import Table
9from rich.text import Text
11from documents.management.commands.base import PaperlessCommand
12from documents.models import Document
15@dataclasses.dataclass(frozen=True, slots=True)
16class _WorkPackage:
17 pk_a: int
18 content_a: str
19 pk_b: int
20 content_b: str
21 score_cutoff: float
24@dataclasses.dataclass(frozen=True, slots=True)
25class _WorkResult:
26 doc_one_pk: int
27 doc_two_pk: int
28 ratio: float
31def _process_and_match(work: _WorkPackage) -> _WorkResult:
32 """
33 Process document content and compute the fuzzy ratio.
34 score_cutoff lets rapidfuzz short-circuit when the score cannot reach the threshold.
35 """
36 first_string = rapidfuzz.utils.default_process(work.content_a)
37 second_string = rapidfuzz.utils.default_process(work.content_b)
38 ratio = rapidfuzz.fuzz.ratio(
39 first_string,
40 second_string,
41 score_cutoff=work.score_cutoff,
42 )
43 return _WorkResult(work.pk_a, work.pk_b, ratio)
46class Command(PaperlessCommand):
47 help = "Searches for documents where the content almost matches"
49 supports_progress_bar = True
50 supports_multiprocessing = True
52 def add_arguments(self, parser):
53 super().add_arguments(parser)
54 parser.add_argument(
55 "--ratio",
56 default=85.0,
57 type=float,
58 help="Ratio to consider documents a match (0.0 - 100.0)",
59 )
60 parser.add_argument(
61 "--delete",
62 default=False,
63 action="store_true",
64 help="If set, one document of matches above the ratio WILL BE DELETED",
65 )
66 parser.add_argument(
67 "--yes",
68 default=False,
69 action="store_true",
70 help="Skip the confirmation prompt when used with --delete",
71 )
72 parser.add_argument(
73 "--url",
74 default=None,
75 type=str,
76 help=(
77 "Base URL of the Paperless instance (e.g. "
78 "http://localhost:8000 or https://paperless.local). If set, matched "
79 "documents are shown as clickable (usually ctrl+click) links to "
80 "<url>/documents/<id>/details instead of by title."
81 ),
82 )
84 def _render_results(
85 self,
86 matches: list[_WorkResult],
87 *,
88 opt_ratio: float,
89 do_delete: bool,
90 base_url: str | None = None,
91 ) -> list[int]:
92 """Render match results as a Rich table. Returns list of PKs to delete."""
93 if not matches:
94 self.console.print(
95 Panel(
96 "[green]No duplicate documents found.[/green]",
97 title="Fuzzy Match",
98 border_style="green",
99 ),
100 )
101 return []
103 # Fetch titles for matched documents in a single query, unless we're
104 # going to show URLs instead.
105 titles: dict[int, str] = {}
106 if not base_url:
107 all_pks = {pk for m in matches for pk in (m.doc_one_pk, m.doc_two_pk)}
108 titles = dict(
109 Document.objects.filter(pk__in=all_pks)
110 .only("pk", "title")
111 .values_list("pk", "title"),
112 )
114 def _cell(pk: int) -> str:
115 if base_url:
116 doc_url = f"{base_url.rstrip('/')}/documents/{pk}/details"
117 return f"[link={doc_url}]{doc_url}[/link]"
118 return f"[dim]#{pk}[/dim] {titles.get(pk, 'Unknown')}"
120 table = Table(
121 title=f"Fuzzy Matches (threshold: {opt_ratio:.1f}%)",
122 show_lines=True,
123 title_style="bold",
124 )
125 table.add_column("#", style="dim", width=4, no_wrap=True)
126 table.add_column("Document A", min_width=24)
127 table.add_column("Document B", min_width=24)
128 table.add_column("Similarity", width=11, justify="right")
130 maybe_delete_ids: list[int] = []
132 for i, match_result in enumerate(matches, 1):
133 pk_a = match_result.doc_one_pk
134 pk_b = match_result.doc_two_pk
135 ratio = match_result.ratio
137 if ratio >= 97.0:
138 ratio_style = "bold red"
139 elif ratio >= 92.0:
140 ratio_style = "red"
141 elif ratio >= 88.0:
142 ratio_style = "yellow"
143 else:
144 ratio_style = "dim"
146 table.add_row(
147 str(i),
148 _cell(pk_a),
149 _cell(pk_b),
150 Text(f"{ratio:.1f}%", style=ratio_style),
151 )
152 maybe_delete_ids.append(pk_b)
154 self.console.print(table)
156 summary = f"Found [bold]{len(matches)}[/bold] matching pair(s)."
157 if do_delete:
158 summary += f" [yellow]{len(maybe_delete_ids)}[/yellow] document(s) will be deleted."
159 self.console.print(summary)
161 return maybe_delete_ids
163 def handle(self, *args, **options):
164 RATIO_MIN: Final[float] = 0.0
165 RATIO_MAX: Final[float] = 100.0
167 opt_ratio = options["ratio"]
169 if opt_ratio < RATIO_MIN or opt_ratio > RATIO_MAX:
170 raise CommandError("The ratio must be between 0 and 100")
172 if options["delete"]:
173 self.console.print(
174 Panel(
175 "[bold yellow]WARNING:[/bold yellow] This run is configured to delete"
176 " documents. One document from each matched pair WILL BE PERMANENTLY DELETED.",
177 title="Delete Mode",
178 border_style="red",
179 ),
180 )
182 # Load only the fields we need -- avoids fetching title, archive_checksum, etc.
183 slim_docs: list[tuple[int, str]] = list(
184 Document.objects.only("id", "content")
185 .order_by("id")
186 .values_list("id", "content"),
187 )
189 # combinations() generates each unique pair exactly once -- no checked_pairs set needed.
190 # The total is computed cheaply so the progress bar can start immediately without
191 # materialising all pairs up front (n*(n-1)/2 can be hundreds of thousands).
192 n = len(slim_docs)
193 total_pairs = n * (n - 1) // 2
195 def _work_gen():
196 for (pk_a, ca), (pk_b, cb) in combinations(slim_docs, 2):
197 if ca.strip() and cb.strip():
198 yield _WorkPackage(pk_a, ca, pk_b, cb, opt_ratio)
200 def _iter_matches():
201 if self.process_count == 1:
202 for work in self.track(
203 _work_gen(),
204 description="Matching...",
205 total=total_pairs,
206 ):
207 result = _process_and_match(work)
208 if result.ratio >= opt_ratio:
209 yield result
210 else: # pragma: no cover
211 work_pkgs = list(_work_gen())
212 for proc_result in self.process_parallel(
213 _process_and_match,
214 work_pkgs,
215 description="Matching...",
216 ):
217 if proc_result.error:
218 self.console.print(
219 f"[red]Failed: {proc_result.error}[/red]",
220 )
221 elif (
222 proc_result.result is not None
223 and proc_result.result.ratio >= opt_ratio
224 ):
225 yield proc_result.result
227 matches = sorted(_iter_matches(), key=lambda m: m.ratio, reverse=True)
228 maybe_delete_ids = self._render_results(
229 matches,
230 opt_ratio=opt_ratio,
231 do_delete=options["delete"],
232 base_url=options["url"],
233 )
235 if options["delete"] and maybe_delete_ids:
236 confirmed = options["yes"]
237 if not confirmed:
238 self.console.print(
239 f"\nDelete [bold]{len(maybe_delete_ids)}[/bold] document(s)? "
240 "[bold]\\[y/N][/bold] ",
241 end="",
242 )
243 answer = input().strip().lower()
244 confirmed = answer in {"y", "yes"}
246 if confirmed:
247 self.console.print(
248 f"[red]Deleting {len(maybe_delete_ids)} document(s)...[/red]",
249 )
250 Document.objects.filter(pk__in=maybe_delete_ids).delete()
251 self.console.print("[green]Done.[/green]")
252 else:
253 self.console.print("[yellow]Deletion cancelled.[/yellow]")