Coverage for documents/management/commands/document_fuzzy_match.py: 0%

107 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1import dataclasses 

2from itertools import combinations 

3from typing import Final 

4 

5import rapidfuzz 

6from django.core.management import CommandError 

7from rich.panel import Panel 

8from rich.table import Table 

9from rich.text import Text 

10 

11from documents.management.commands.base import PaperlessCommand 

12from documents.models import Document 

13 

14 

15@dataclasses.dataclass(frozen=True, slots=True) 

16class _WorkPackage: 

17 pk_a: int 

18 content_a: str 

19 pk_b: int 

20 content_b: str 

21 score_cutoff: float 

22 

23 

24@dataclasses.dataclass(frozen=True, slots=True) 

25class _WorkResult: 

26 doc_one_pk: int 

27 doc_two_pk: int 

28 ratio: float 

29 

30 

31def _process_and_match(work: _WorkPackage) -> _WorkResult: 

32 """ 

33 Process document content and compute the fuzzy ratio. 

34 score_cutoff lets rapidfuzz short-circuit when the score cannot reach the threshold. 

35 """ 

36 first_string = rapidfuzz.utils.default_process(work.content_a) 

37 second_string = rapidfuzz.utils.default_process(work.content_b) 

38 ratio = rapidfuzz.fuzz.ratio( 

39 first_string, 

40 second_string, 

41 score_cutoff=work.score_cutoff, 

42 ) 

43 return _WorkResult(work.pk_a, work.pk_b, ratio) 

44 

45 

46class Command(PaperlessCommand): 

47 help = "Searches for documents where the content almost matches" 

48 

49 supports_progress_bar = True 

50 supports_multiprocessing = True 

51 

52 def add_arguments(self, parser): 

53 super().add_arguments(parser) 

54 parser.add_argument( 

55 "--ratio", 

56 default=85.0, 

57 type=float, 

58 help="Ratio to consider documents a match (0.0 - 100.0)", 

59 ) 

60 parser.add_argument( 

61 "--delete", 

62 default=False, 

63 action="store_true", 

64 help="If set, one document of matches above the ratio WILL BE DELETED", 

65 ) 

66 parser.add_argument( 

67 "--yes", 

68 default=False, 

69 action="store_true", 

70 help="Skip the confirmation prompt when used with --delete", 

71 ) 

72 parser.add_argument( 

73 "--url", 

74 default=None, 

75 type=str, 

76 help=( 

77 "Base URL of the Paperless instance (e.g. " 

78 "http://localhost:8000 or https://paperless.local). If set, matched " 

79 "documents are shown as clickable (usually ctrl+click) links to " 

80 "<url>/documents/<id>/details instead of by title." 

81 ), 

82 ) 

83 

84 def _render_results( 

85 self, 

86 matches: list[_WorkResult], 

87 *, 

88 opt_ratio: float, 

89 do_delete: bool, 

90 base_url: str | None = None, 

91 ) -> list[int]: 

92 """Render match results as a Rich table. Returns list of PKs to delete.""" 

93 if not matches: 

94 self.console.print( 

95 Panel( 

96 "[green]No duplicate documents found.[/green]", 

97 title="Fuzzy Match", 

98 border_style="green", 

99 ), 

100 ) 

101 return [] 

102 

103 # Fetch titles for matched documents in a single query, unless we're 

104 # going to show URLs instead. 

105 titles: dict[int, str] = {} 

106 if not base_url: 

107 all_pks = {pk for m in matches for pk in (m.doc_one_pk, m.doc_two_pk)} 

108 titles = dict( 

109 Document.objects.filter(pk__in=all_pks) 

110 .only("pk", "title") 

111 .values_list("pk", "title"), 

112 ) 

113 

114 def _cell(pk: int) -> str: 

115 if base_url: 

116 doc_url = f"{base_url.rstrip('/')}/documents/{pk}/details" 

117 return f"[link={doc_url}]{doc_url}[/link]" 

118 return f"[dim]#{pk}[/dim] {titles.get(pk, 'Unknown')}" 

119 

120 table = Table( 

121 title=f"Fuzzy Matches (threshold: {opt_ratio:.1f}%)", 

122 show_lines=True, 

123 title_style="bold", 

124 ) 

125 table.add_column("#", style="dim", width=4, no_wrap=True) 

126 table.add_column("Document A", min_width=24) 

127 table.add_column("Document B", min_width=24) 

128 table.add_column("Similarity", width=11, justify="right") 

129 

130 maybe_delete_ids: list[int] = [] 

131 

132 for i, match_result in enumerate(matches, 1): 

133 pk_a = match_result.doc_one_pk 

134 pk_b = match_result.doc_two_pk 

135 ratio = match_result.ratio 

136 

137 if ratio >= 97.0: 

138 ratio_style = "bold red" 

139 elif ratio >= 92.0: 

140 ratio_style = "red" 

141 elif ratio >= 88.0: 

142 ratio_style = "yellow" 

143 else: 

144 ratio_style = "dim" 

145 

146 table.add_row( 

147 str(i), 

148 _cell(pk_a), 

149 _cell(pk_b), 

150 Text(f"{ratio:.1f}%", style=ratio_style), 

151 ) 

152 maybe_delete_ids.append(pk_b) 

153 

154 self.console.print(table) 

155 

156 summary = f"Found [bold]{len(matches)}[/bold] matching pair(s)." 

157 if do_delete: 

158 summary += f" [yellow]{len(maybe_delete_ids)}[/yellow] document(s) will be deleted." 

159 self.console.print(summary) 

160 

161 return maybe_delete_ids 

162 

163 def handle(self, *args, **options): 

164 RATIO_MIN: Final[float] = 0.0 

165 RATIO_MAX: Final[float] = 100.0 

166 

167 opt_ratio = options["ratio"] 

168 

169 if opt_ratio < RATIO_MIN or opt_ratio > RATIO_MAX: 

170 raise CommandError("The ratio must be between 0 and 100") 

171 

172 if options["delete"]: 

173 self.console.print( 

174 Panel( 

175 "[bold yellow]WARNING:[/bold yellow] This run is configured to delete" 

176 " documents. One document from each matched pair WILL BE PERMANENTLY DELETED.", 

177 title="Delete Mode", 

178 border_style="red", 

179 ), 

180 ) 

181 

182 # Load only the fields we need -- avoids fetching title, archive_checksum, etc. 

183 slim_docs: list[tuple[int, str]] = list( 

184 Document.objects.only("id", "content") 

185 .order_by("id") 

186 .values_list("id", "content"), 

187 ) 

188 

189 # combinations() generates each unique pair exactly once -- no checked_pairs set needed. 

190 # The total is computed cheaply so the progress bar can start immediately without 

191 # materialising all pairs up front (n*(n-1)/2 can be hundreds of thousands). 

192 n = len(slim_docs) 

193 total_pairs = n * (n - 1) // 2 

194 

195 def _work_gen(): 

196 for (pk_a, ca), (pk_b, cb) in combinations(slim_docs, 2): 

197 if ca.strip() and cb.strip(): 

198 yield _WorkPackage(pk_a, ca, pk_b, cb, opt_ratio) 

199 

200 def _iter_matches(): 

201 if self.process_count == 1: 

202 for work in self.track( 

203 _work_gen(), 

204 description="Matching...", 

205 total=total_pairs, 

206 ): 

207 result = _process_and_match(work) 

208 if result.ratio >= opt_ratio: 

209 yield result 

210 else: # pragma: no cover 

211 work_pkgs = list(_work_gen()) 

212 for proc_result in self.process_parallel( 

213 _process_and_match, 

214 work_pkgs, 

215 description="Matching...", 

216 ): 

217 if proc_result.error: 

218 self.console.print( 

219 f"[red]Failed: {proc_result.error}[/red]", 

220 ) 

221 elif ( 

222 proc_result.result is not None 

223 and proc_result.result.ratio >= opt_ratio 

224 ): 

225 yield proc_result.result 

226 

227 matches = sorted(_iter_matches(), key=lambda m: m.ratio, reverse=True) 

228 maybe_delete_ids = self._render_results( 

229 matches, 

230 opt_ratio=opt_ratio, 

231 do_delete=options["delete"], 

232 base_url=options["url"], 

233 ) 

234 

235 if options["delete"] and maybe_delete_ids: 

236 confirmed = options["yes"] 

237 if not confirmed: 

238 self.console.print( 

239 f"\nDelete [bold]{len(maybe_delete_ids)}[/bold] document(s)? " 

240 "[bold]\\[y/N][/bold] ", 

241 end="", 

242 ) 

243 answer = input().strip().lower() 

244 confirmed = answer in {"y", "yes"} 

245 

246 if confirmed: 

247 self.console.print( 

248 f"[red]Deleting {len(maybe_delete_ids)} document(s)...[/red]", 

249 ) 

250 Document.objects.filter(pk__in=maybe_delete_ids).delete() 

251 self.console.print("[green]Done.[/green]") 

252 else: 

253 self.console.print("[yellow]Deletion cancelled.[/yellow]")