Coverage for documents/management/commands/document_retagger.py: 0%
160 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import logging
4from dataclasses import dataclass
5from dataclasses import field
6from typing import TYPE_CHECKING
8from rich.table import Table
9from rich.text import Text
11from documents.classifier import load_classifier
12from documents.management.commands.base import PaperlessCommand
13from documents.models import Document
14from documents.signals.handlers import set_correspondent
15from documents.signals.handlers import set_document_type
16from documents.signals.handlers import set_storage_path
17from documents.signals.handlers import set_tags
19if TYPE_CHECKING:
20 from rich.console import RenderableType
22 from documents.models import Correspondent
23 from documents.models import DocumentType
24 from documents.models import StoragePath
25 from documents.models import Tag
27logger = logging.getLogger("paperless.management.retagger")
30@dataclass(slots=True)
31class RetaggerStats:
32 """Cumulative counters updated as the retagger processes documents.
34 Mutable by design -- fields are incremented in the processing loop.
35 slots=True reduces per-instance memory overhead and speeds attribute access.
36 """
38 correspondents: int = 0
39 document_types: int = 0
40 tags_added: int = 0
41 tags_removed: int = 0
42 storage_paths: int = 0
43 documents_processed: int = 0
46@dataclass(slots=True)
47class DocumentSuggestion:
48 """Buffered classifier suggestions for a single document (suggest mode only).
50 Mutable by design -- fields are assigned incrementally as each setter runs.
51 """
53 document: Document
54 correspondent: Correspondent | None = None
55 document_type: DocumentType | None = None
56 tags_to_add: frozenset[Tag] = field(default_factory=frozenset)
57 tags_to_remove: frozenset[Tag] = field(default_factory=frozenset)
58 storage_path: StoragePath | None = None
60 @property
61 def has_suggestions(self) -> bool:
62 return bool(
63 self.correspondent is not None
64 or self.document_type is not None
65 or self.tags_to_add
66 or self.tags_to_remove
67 or self.storage_path is not None,
68 )
71def _build_stats_table(stats: RetaggerStats, *, suggest: bool) -> Table:
72 """
73 Build the live-updating stats table shown below the progress bar.
75 In suggest mode the labels read "would set / would add" to make clear
76 that nothing has been written to the database.
77 """
78 table = Table(box=None, padding=(0, 2), show_header=True, header_style="bold")
80 table.add_column("Documents")
81 table.add_column("Correspondents")
82 table.add_column("Doc Types")
83 table.add_column("Tags (+)")
84 table.add_column("Tags (-)")
85 table.add_column("Storage Paths")
87 verb = "would set" if suggest else "set"
89 table.add_row(
90 str(stats.documents_processed),
91 f"{stats.correspondents} {verb}",
92 f"{stats.document_types} {verb}",
93 f"+{stats.tags_added}",
94 f"-{stats.tags_removed}",
95 f"{stats.storage_paths} {verb}",
96 )
98 return table
101def _build_suggestion_table(
102 suggestions: list[DocumentSuggestion],
103 base_url: str | None,
104) -> Table:
105 """
106 Build the final suggestion table printed after the progress bar completes.
108 Only documents with at least one suggestion are included.
109 """
110 table = Table(
111 title="Suggested Changes",
112 show_header=True,
113 header_style="bold cyan",
114 show_lines=True,
115 )
117 table.add_column("Document", style="bold", no_wrap=False, min_width=20)
118 table.add_column("Correspondent")
119 table.add_column("Doc Type")
120 table.add_column("Tags")
121 table.add_column("Storage Path")
123 for suggestion in suggestions:
124 if not suggestion.has_suggestions:
125 continue
127 doc = suggestion.document
129 if base_url:
130 doc_cell = Text()
131 doc_cell.append(str(doc))
132 doc_cell.append(f"\n{base_url}/documents/{doc.pk}", style="dim")
133 else:
134 doc_cell = Text(f"{doc} [{doc.pk}]")
136 tag_parts: list[str] = []
137 for tag in sorted(suggestion.tags_to_add, key=lambda t: t.name):
138 tag_parts.append(f"[green]+{tag.name}[/green]")
139 for tag in sorted(suggestion.tags_to_remove, key=lambda t: t.name):
140 tag_parts.append(f"[red]-{tag.name}[/red]")
141 tag_cell = Text.from_markup(", ".join(tag_parts)) if tag_parts else Text("-")
143 table.add_row(
144 doc_cell,
145 str(suggestion.correspondent) if suggestion.correspondent else "-",
146 str(suggestion.document_type) if suggestion.document_type else "-",
147 tag_cell,
148 str(suggestion.storage_path) if suggestion.storage_path else "-",
149 )
151 return table
154def _build_summary_table(stats: RetaggerStats) -> Table:
155 """Build the final applied-changes summary table."""
156 table = Table(
157 title="Retagger Summary",
158 show_header=True,
159 header_style="bold cyan",
160 )
162 table.add_column("Metric", style="bold")
163 table.add_column("Count", justify="right")
165 table.add_row("Documents processed", str(stats.documents_processed))
166 table.add_row("Correspondents set", str(stats.correspondents))
167 table.add_row("Document types set", str(stats.document_types))
168 table.add_row("Tags added", str(stats.tags_added))
169 table.add_row("Tags removed", str(stats.tags_removed))
170 table.add_row("Storage paths set", str(stats.storage_paths))
172 return table
175class Command(PaperlessCommand):
176 help = (
177 "Using the current classification model, assigns correspondents, tags "
178 "and document types to all documents, effectively allowing you to "
179 "back-tag all previously indexed documents with metadata created (or "
180 "modified) after their initial import."
181 )
183 supports_progress_bar = True
184 supports_multiprocessing = False
186 def add_arguments(self, parser) -> None:
187 super().add_arguments(parser)
188 parser.add_argument("-c", "--correspondent", default=False, action="store_true")
189 parser.add_argument("-T", "--tags", default=False, action="store_true")
190 parser.add_argument("-t", "--document_type", default=False, action="store_true")
191 parser.add_argument("-s", "--storage_path", default=False, action="store_true")
192 parser.add_argument("-i", "--inbox-only", default=False, action="store_true")
193 parser.add_argument(
194 "--use-first",
195 default=False,
196 action="store_true",
197 help=(
198 "By default this command will not try to assign a correspondent "
199 "if more than one matches the document. Use this flag to pick "
200 "the first match instead."
201 ),
202 )
203 parser.add_argument(
204 "-f",
205 "--overwrite",
206 default=False,
207 action="store_true",
208 help=(
209 "Overwrite any previously set correspondent, document type, and "
210 "remove tags that no longer match due to changed rules."
211 ),
212 )
213 parser.add_argument(
214 "--suggest",
215 default=False,
216 action="store_true",
217 help="Show what would be changed without applying anything.",
218 )
219 parser.add_argument(
220 "--base-url",
221 help="Base URL used to build document links in suggest output.",
222 )
223 parser.add_argument(
224 "--id-range",
225 help="Restrict retagging to documents within this ID range (inclusive).",
226 nargs=2,
227 type=int,
228 )
230 def handle(self, *args, **options) -> None:
231 suggest: bool = options["suggest"]
232 overwrite: bool = options["overwrite"]
233 use_first: bool = options["use_first"]
234 base_url: str | None = options["base_url"]
236 do_correspondent: bool = options["correspondent"]
237 do_document_type: bool = options["document_type"]
238 do_tags: bool = options["tags"]
239 do_storage_path: bool = options["storage_path"]
241 if not any([do_correspondent, do_document_type, do_tags, do_storage_path]):
242 self.console.print(
243 "[yellow]No classifier targets specified. "
244 "Use -c, -T, -t, or -s to select what to retag.[/yellow]",
245 )
246 return
248 if options["inbox_only"]:
249 queryset = Document.objects.filter(tags__is_inbox_tag=True)
250 else:
251 queryset = Document.objects.all()
253 if options["id_range"]:
254 lo, hi = options["id_range"]
255 queryset = queryset.filter(id__range=(lo, hi))
257 documents = queryset.distinct()
258 classifier = load_classifier()
260 stats = RetaggerStats()
261 suggestions: list[DocumentSuggestion] = []
263 def render_stats() -> RenderableType:
264 return _build_stats_table(stats, suggest=suggest)
266 with self.buffered_logging(
267 "paperless",
268 "paperless.handlers",
269 "documents",
270 ) as log_buf:
271 for document in self.track_with_stats(
272 documents,
273 description="Retagging...",
274 stats_renderer=render_stats,
275 ):
276 suggestion = DocumentSuggestion(document=document)
278 if do_correspondent:
279 correspondent = set_correspondent(
280 None,
281 document,
282 classifier=classifier,
283 replace=overwrite,
284 use_first=use_first,
285 dry_run=suggest,
286 )
287 if correspondent is not None:
288 stats.correspondents += 1
289 suggestion.correspondent = correspondent
291 if do_document_type:
292 document_type = set_document_type(
293 None,
294 document,
295 classifier=classifier,
296 replace=overwrite,
297 use_first=use_first,
298 dry_run=suggest,
299 )
300 if document_type is not None:
301 stats.document_types += 1
302 suggestion.document_type = document_type
304 if do_tags:
305 tags_to_add, tags_to_remove = set_tags(
306 None,
307 document,
308 classifier=classifier,
309 replace=overwrite,
310 dry_run=suggest,
311 )
312 stats.tags_added += len(tags_to_add)
313 stats.tags_removed += len(tags_to_remove)
314 suggestion.tags_to_add = frozenset(tags_to_add)
315 suggestion.tags_to_remove = frozenset(tags_to_remove)
317 if do_storage_path:
318 storage_path = set_storage_path(
319 None,
320 document,
321 classifier=classifier,
322 replace=overwrite,
323 use_first=use_first,
324 dry_run=suggest,
325 )
326 if storage_path is not None:
327 stats.storage_paths += 1
328 suggestion.storage_path = storage_path
330 stats.documents_processed += 1
332 if suggest:
333 suggestions.append(suggestion)
335 # Post-loop output
336 if suggest:
337 visible = [s for s in suggestions if s.has_suggestions]
338 if visible:
339 self.console.print(_build_suggestion_table(visible, base_url))
340 else:
341 self.console.print("[green]No changes suggested.[/green]")
342 else:
343 self.console.print(_build_summary_table(stats))
345 log_buf.render(self.console, min_level=logging.INFO, title="Retagger Log")