Coverage for documents/management/commands/document_retagger.py: 0%

160 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import logging 

4from dataclasses import dataclass 

5from dataclasses import field 

6from typing import TYPE_CHECKING 

7 

8from rich.table import Table 

9from rich.text import Text 

10 

11from documents.classifier import load_classifier 

12from documents.management.commands.base import PaperlessCommand 

13from documents.models import Document 

14from documents.signals.handlers import set_correspondent 

15from documents.signals.handlers import set_document_type 

16from documents.signals.handlers import set_storage_path 

17from documents.signals.handlers import set_tags 

18 

19if TYPE_CHECKING: 

20 from rich.console import RenderableType 

21 

22 from documents.models import Correspondent 

23 from documents.models import DocumentType 

24 from documents.models import StoragePath 

25 from documents.models import Tag 

26 

27logger = logging.getLogger("paperless.management.retagger") 

28 

29 

30@dataclass(slots=True) 

31class RetaggerStats: 

32 """Cumulative counters updated as the retagger processes documents. 

33 

34 Mutable by design -- fields are incremented in the processing loop. 

35 slots=True reduces per-instance memory overhead and speeds attribute access. 

36 """ 

37 

38 correspondents: int = 0 

39 document_types: int = 0 

40 tags_added: int = 0 

41 tags_removed: int = 0 

42 storage_paths: int = 0 

43 documents_processed: int = 0 

44 

45 

46@dataclass(slots=True) 

47class DocumentSuggestion: 

48 """Buffered classifier suggestions for a single document (suggest mode only). 

49 

50 Mutable by design -- fields are assigned incrementally as each setter runs. 

51 """ 

52 

53 document: Document 

54 correspondent: Correspondent | None = None 

55 document_type: DocumentType | None = None 

56 tags_to_add: frozenset[Tag] = field(default_factory=frozenset) 

57 tags_to_remove: frozenset[Tag] = field(default_factory=frozenset) 

58 storage_path: StoragePath | None = None 

59 

60 @property 

61 def has_suggestions(self) -> bool: 

62 return bool( 

63 self.correspondent is not None 

64 or self.document_type is not None 

65 or self.tags_to_add 

66 or self.tags_to_remove 

67 or self.storage_path is not None, 

68 ) 

69 

70 

71def _build_stats_table(stats: RetaggerStats, *, suggest: bool) -> Table: 

72 """ 

73 Build the live-updating stats table shown below the progress bar. 

74 

75 In suggest mode the labels read "would set / would add" to make clear 

76 that nothing has been written to the database. 

77 """ 

78 table = Table(box=None, padding=(0, 2), show_header=True, header_style="bold") 

79 

80 table.add_column("Documents") 

81 table.add_column("Correspondents") 

82 table.add_column("Doc Types") 

83 table.add_column("Tags (+)") 

84 table.add_column("Tags (-)") 

85 table.add_column("Storage Paths") 

86 

87 verb = "would set" if suggest else "set" 

88 

89 table.add_row( 

90 str(stats.documents_processed), 

91 f"{stats.correspondents} {verb}", 

92 f"{stats.document_types} {verb}", 

93 f"+{stats.tags_added}", 

94 f"-{stats.tags_removed}", 

95 f"{stats.storage_paths} {verb}", 

96 ) 

97 

98 return table 

99 

100 

101def _build_suggestion_table( 

102 suggestions: list[DocumentSuggestion], 

103 base_url: str | None, 

104) -> Table: 

105 """ 

106 Build the final suggestion table printed after the progress bar completes. 

107 

108 Only documents with at least one suggestion are included. 

109 """ 

110 table = Table( 

111 title="Suggested Changes", 

112 show_header=True, 

113 header_style="bold cyan", 

114 show_lines=True, 

115 ) 

116 

117 table.add_column("Document", style="bold", no_wrap=False, min_width=20) 

118 table.add_column("Correspondent") 

119 table.add_column("Doc Type") 

120 table.add_column("Tags") 

121 table.add_column("Storage Path") 

122 

123 for suggestion in suggestions: 

124 if not suggestion.has_suggestions: 

125 continue 

126 

127 doc = suggestion.document 

128 

129 if base_url: 

130 doc_cell = Text() 

131 doc_cell.append(str(doc)) 

132 doc_cell.append(f"\n{base_url}/documents/{doc.pk}", style="dim") 

133 else: 

134 doc_cell = Text(f"{doc} [{doc.pk}]") 

135 

136 tag_parts: list[str] = [] 

137 for tag in sorted(suggestion.tags_to_add, key=lambda t: t.name): 

138 tag_parts.append(f"[green]+{tag.name}[/green]") 

139 for tag in sorted(suggestion.tags_to_remove, key=lambda t: t.name): 

140 tag_parts.append(f"[red]-{tag.name}[/red]") 

141 tag_cell = Text.from_markup(", ".join(tag_parts)) if tag_parts else Text("-") 

142 

143 table.add_row( 

144 doc_cell, 

145 str(suggestion.correspondent) if suggestion.correspondent else "-", 

146 str(suggestion.document_type) if suggestion.document_type else "-", 

147 tag_cell, 

148 str(suggestion.storage_path) if suggestion.storage_path else "-", 

149 ) 

150 

151 return table 

152 

153 

154def _build_summary_table(stats: RetaggerStats) -> Table: 

155 """Build the final applied-changes summary table.""" 

156 table = Table( 

157 title="Retagger Summary", 

158 show_header=True, 

159 header_style="bold cyan", 

160 ) 

161 

162 table.add_column("Metric", style="bold") 

163 table.add_column("Count", justify="right") 

164 

165 table.add_row("Documents processed", str(stats.documents_processed)) 

166 table.add_row("Correspondents set", str(stats.correspondents)) 

167 table.add_row("Document types set", str(stats.document_types)) 

168 table.add_row("Tags added", str(stats.tags_added)) 

169 table.add_row("Tags removed", str(stats.tags_removed)) 

170 table.add_row("Storage paths set", str(stats.storage_paths)) 

171 

172 return table 

173 

174 

175class Command(PaperlessCommand): 

176 help = ( 

177 "Using the current classification model, assigns correspondents, tags " 

178 "and document types to all documents, effectively allowing you to " 

179 "back-tag all previously indexed documents with metadata created (or " 

180 "modified) after their initial import." 

181 ) 

182 

183 supports_progress_bar = True 

184 supports_multiprocessing = False 

185 

186 def add_arguments(self, parser) -> None: 

187 super().add_arguments(parser) 

188 parser.add_argument("-c", "--correspondent", default=False, action="store_true") 

189 parser.add_argument("-T", "--tags", default=False, action="store_true") 

190 parser.add_argument("-t", "--document_type", default=False, action="store_true") 

191 parser.add_argument("-s", "--storage_path", default=False, action="store_true") 

192 parser.add_argument("-i", "--inbox-only", default=False, action="store_true") 

193 parser.add_argument( 

194 "--use-first", 

195 default=False, 

196 action="store_true", 

197 help=( 

198 "By default this command will not try to assign a correspondent " 

199 "if more than one matches the document. Use this flag to pick " 

200 "the first match instead." 

201 ), 

202 ) 

203 parser.add_argument( 

204 "-f", 

205 "--overwrite", 

206 default=False, 

207 action="store_true", 

208 help=( 

209 "Overwrite any previously set correspondent, document type, and " 

210 "remove tags that no longer match due to changed rules." 

211 ), 

212 ) 

213 parser.add_argument( 

214 "--suggest", 

215 default=False, 

216 action="store_true", 

217 help="Show what would be changed without applying anything.", 

218 ) 

219 parser.add_argument( 

220 "--base-url", 

221 help="Base URL used to build document links in suggest output.", 

222 ) 

223 parser.add_argument( 

224 "--id-range", 

225 help="Restrict retagging to documents within this ID range (inclusive).", 

226 nargs=2, 

227 type=int, 

228 ) 

229 

230 def handle(self, *args, **options) -> None: 

231 suggest: bool = options["suggest"] 

232 overwrite: bool = options["overwrite"] 

233 use_first: bool = options["use_first"] 

234 base_url: str | None = options["base_url"] 

235 

236 do_correspondent: bool = options["correspondent"] 

237 do_document_type: bool = options["document_type"] 

238 do_tags: bool = options["tags"] 

239 do_storage_path: bool = options["storage_path"] 

240 

241 if not any([do_correspondent, do_document_type, do_tags, do_storage_path]): 

242 self.console.print( 

243 "[yellow]No classifier targets specified. " 

244 "Use -c, -T, -t, or -s to select what to retag.[/yellow]", 

245 ) 

246 return 

247 

248 if options["inbox_only"]: 

249 queryset = Document.objects.filter(tags__is_inbox_tag=True) 

250 else: 

251 queryset = Document.objects.all() 

252 

253 if options["id_range"]: 

254 lo, hi = options["id_range"] 

255 queryset = queryset.filter(id__range=(lo, hi)) 

256 

257 documents = queryset.distinct() 

258 classifier = load_classifier() 

259 

260 stats = RetaggerStats() 

261 suggestions: list[DocumentSuggestion] = [] 

262 

263 def render_stats() -> RenderableType: 

264 return _build_stats_table(stats, suggest=suggest) 

265 

266 with self.buffered_logging( 

267 "paperless", 

268 "paperless.handlers", 

269 "documents", 

270 ) as log_buf: 

271 for document in self.track_with_stats( 

272 documents, 

273 description="Retagging...", 

274 stats_renderer=render_stats, 

275 ): 

276 suggestion = DocumentSuggestion(document=document) 

277 

278 if do_correspondent: 

279 correspondent = set_correspondent( 

280 None, 

281 document, 

282 classifier=classifier, 

283 replace=overwrite, 

284 use_first=use_first, 

285 dry_run=suggest, 

286 ) 

287 if correspondent is not None: 

288 stats.correspondents += 1 

289 suggestion.correspondent = correspondent 

290 

291 if do_document_type: 

292 document_type = set_document_type( 

293 None, 

294 document, 

295 classifier=classifier, 

296 replace=overwrite, 

297 use_first=use_first, 

298 dry_run=suggest, 

299 ) 

300 if document_type is not None: 

301 stats.document_types += 1 

302 suggestion.document_type = document_type 

303 

304 if do_tags: 

305 tags_to_add, tags_to_remove = set_tags( 

306 None, 

307 document, 

308 classifier=classifier, 

309 replace=overwrite, 

310 dry_run=suggest, 

311 ) 

312 stats.tags_added += len(tags_to_add) 

313 stats.tags_removed += len(tags_to_remove) 

314 suggestion.tags_to_add = frozenset(tags_to_add) 

315 suggestion.tags_to_remove = frozenset(tags_to_remove) 

316 

317 if do_storage_path: 

318 storage_path = set_storage_path( 

319 None, 

320 document, 

321 classifier=classifier, 

322 replace=overwrite, 

323 use_first=use_first, 

324 dry_run=suggest, 

325 ) 

326 if storage_path is not None: 

327 stats.storage_paths += 1 

328 suggestion.storage_path = storage_path 

329 

330 stats.documents_processed += 1 

331 

332 if suggest: 

333 suggestions.append(suggestion) 

334 

335 # Post-loop output 

336 if suggest: 

337 visible = [s for s in suggestions if s.has_suggestions] 

338 if visible: 

339 self.console.print(_build_suggestion_table(visible, base_url)) 

340 else: 

341 self.console.print("[green]No changes suggested.[/green]") 

342 else: 

343 self.console.print(_build_summary_table(stats)) 

344 

345 log_buf.render(self.console, min_level=logging.INFO, title="Retagger Log")