Coverage for documents/management/commands/document_exporter.py: 0%
258 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1import os
2from itertools import islice
3from pathlib import Path
4from typing import TYPE_CHECKING
5from typing import Any
7from allauth.mfa.models import Authenticator
8from allauth.socialaccount.models import SocialAccount
9from allauth.socialaccount.models import SocialApp
10from allauth.socialaccount.models import SocialToken
11from django.conf import settings
12from django.contrib.auth.models import Group
13from django.contrib.auth.models import Permission
14from django.contrib.auth.models import User
15from django.contrib.contenttypes.models import ContentType
16from django.core import serializers
17from django.core.management.base import CommandError
18from django.db import transaction
19from django.utils import timezone
20from filelock import FileLock
21from guardian.models import GroupObjectPermission
22from guardian.models import UserObjectPermission
24if TYPE_CHECKING:
25 from collections.abc import Generator
27 from django.db.models import QuerySet
29if settings.AUDIT_LOG_ENABLED:
30 from auditlog.models import LogEntry
32from documents.export.compression import COMPRESSION_CHOICES
33from documents.export.compression import COMPRESSION_METHODS
34from documents.export.compression import ZSTD
35from documents.export.compression import compression_available
36from documents.export.compression import level_error
37from documents.export.sinks import DirectoryExportSink
38from documents.export.sinks import ExportSink
39from documents.export.sinks import StreamingManifestWriter
40from documents.export.sinks import ZipExportSink
41from documents.file_handling import generate_filename
42from documents.management.commands.base import PaperlessCommand
43from documents.management.commands.mixins import CryptMixin
44from documents.models import Correspondent
45from documents.models import CustomField
46from documents.models import CustomFieldInstance
47from documents.models import Document
48from documents.models import DocumentBarcode
49from documents.models import DocumentType
50from documents.models import Note
51from documents.models import SavedView
52from documents.models import SavedViewFilterRule
53from documents.models import ShareLink
54from documents.models import ShareLinkBundle
55from documents.models import StoragePath
56from documents.models import Tag
57from documents.models import UiSettings
58from documents.models import Workflow
59from documents.models import WorkflowAction
60from documents.models import WorkflowActionEmail
61from documents.models import WorkflowActionWebhook
62from documents.models import WorkflowTrigger
63from documents.settings import EXPORTER_ARCHIVE_NAME
64from documents.settings import EXPORTER_FILE_NAME
65from documents.settings import EXPORTER_SHARE_LINK_BUNDLE_NAME
66from documents.settings import EXPORTER_THUMBNAIL_NAME
67from documents.utils import QuerySetStream
68from paperless import version
69from paperless.models import ApplicationConfiguration
70from paperless_mail.models import MailAccount
71from paperless_mail.models import MailRule
74def serialize_queryset_batched(
75 queryset: "QuerySet[Any]",
76 *,
77 batch_size: int = 500,
78) -> "Generator[list[dict], None, None]":
79 """Yield batches of serialized records from a QuerySet.
81 Each batch is a list of dicts in Django's Python serialization format.
82 Uses QuerySet.iterator() to avoid loading the full queryset into memory,
83 and islice to collect chunk-sized batches serialized in a single call.
84 """
85 iterator = queryset.iterator(chunk_size=batch_size)
86 while chunk := list(islice(iterator, batch_size)):
87 yield serializers.serialize("python", chunk)
90class Command(CryptMixin, PaperlessCommand):
91 help = (
92 "Decrypt and rename all files in our collection into a given target "
93 "directory. And include a manifest file containing document data for "
94 "easy import."
95 )
97 supports_progress_bar = True
98 supports_multiprocessing = False
100 def add_arguments(self, parser) -> None:
101 super().add_arguments(parser)
102 parser.add_argument("target")
104 parser.add_argument(
105 "-c",
106 "--compare-checksums",
107 default=False,
108 action="store_true",
109 help=(
110 "Compare file checksums when determining whether to export "
111 "a file or not. If not specified, file size and time "
112 "modified is used instead."
113 ),
114 )
116 parser.add_argument(
117 "-cj",
118 "--compare-json",
119 default=False,
120 action="store_true",
121 help=(
122 "Compare json file checksums when determining whether to "
123 "export a json file or not (manifest or metadata). "
124 "If not specified, the file is always exported."
125 ),
126 )
128 parser.add_argument(
129 "-d",
130 "--delete",
131 default=False,
132 action="store_true",
133 help=(
134 "After exporting, delete files in the export directory that "
135 "do not belong to the current export, such as files from "
136 "deleted documents."
137 ),
138 )
140 parser.add_argument(
141 "-f",
142 "--use-filename-format",
143 default=False,
144 action="store_true",
145 help=(
146 "Use PAPERLESS_FILENAME_FORMAT for storing files in the "
147 "export directory, if configured."
148 ),
149 )
151 parser.add_argument(
152 "-na",
153 "--no-archive",
154 default=False,
155 action="store_true",
156 help="Avoid exporting archive files",
157 )
159 parser.add_argument(
160 "-nt",
161 "--no-thumbnail",
162 default=False,
163 action="store_true",
164 help="Avoid exporting thumbnail files",
165 )
167 parser.add_argument(
168 "-p",
169 "--use-folder-prefix",
170 default=False,
171 action="store_true",
172 help=(
173 "Export files in dedicated folders according to their nature: "
174 "archive, originals or thumbnails"
175 ),
176 )
178 parser.add_argument(
179 "-sm",
180 "--split-manifest",
181 default=False,
182 action="store_true",
183 help="Export document information in individual manifest json files.",
184 )
186 parser.add_argument(
187 "-z",
188 "--zip",
189 default=False,
190 action="store_true",
191 help="Export the documents to a zip file in the given directory",
192 )
194 parser.add_argument(
195 "-zn",
196 "--zip-name",
197 default=f"export-{timezone.localdate().isoformat()}",
198 help="Sets the export zip file name",
199 )
201 parser.add_argument(
202 "--zip-compression",
203 choices=COMPRESSION_CHOICES,
204 default=None,
205 help=(
206 "Compression method for the export zip (requires --zip). "
207 "Default: deflated. 'zstd' requires Python 3.14+ on both the "
208 "exporting and importing machine."
209 ),
210 )
212 parser.add_argument(
213 "--zip-compression-level",
214 type=int,
215 default=None,
216 help=(
217 "Compression level for the export zip (requires --zip). "
218 "deflated: 0-9, bzip2: 1-9, zstd: -22..22; ignored for "
219 "stored/lzma."
220 ),
221 )
223 parser.add_argument(
224 "--data-only",
225 default=False,
226 action="store_true",
227 help="If set, only the database will be imported, not files",
228 )
230 parser.add_argument(
231 "--passphrase",
232 help="If provided, is used to encrypt sensitive data in the export",
233 )
235 parser.add_argument(
236 "--batch-size",
237 type=int,
238 default=500,
239 help=(
240 "Number of records to process per batch during serialization. "
241 "Lower values reduce peak memory usage; higher values improve "
242 "throughput. Default: 500."
243 ),
244 )
246 def handle(self, *args, **options) -> None:
247 self.target = Path(options["target"]).resolve()
248 self.split_manifest: bool = options["split_manifest"]
249 self.compare_checksums: bool = options["compare_checksums"]
250 self.compare_json: bool = options["compare_json"]
251 self.use_filename_format: bool = options["use_filename_format"]
252 self.use_folder_prefix: bool = options["use_folder_prefix"]
253 self.delete: bool = options["delete"]
254 self.no_archive: bool = options["no_archive"]
255 self.no_thumbnail: bool = options["no_thumbnail"]
256 self.zip_export: bool = options["zip"]
257 self.data_only: bool = options["data_only"]
258 self.passphrase: str | None = options.get("passphrase")
259 self.batch_size: int = options["batch_size"]
261 self.exported_files: set[str] = set()
263 if self.zip_export and (self.compare_checksums or self.compare_json):
264 raise CommandError(
265 "--compare-checksums and --compare-json have no effect when "
266 "used with --zip",
267 )
269 if not self.target.exists():
270 raise CommandError("That path doesn't exist")
272 if not self.target.is_dir():
273 raise CommandError("That path isn't a directory")
275 if not os.access(self.target, os.W_OK):
276 raise CommandError("That path doesn't appear to be writable")
278 zip_compression: str | None = options["zip_compression"]
279 zip_compression_level: int | None = options["zip_compression_level"]
281 if not self.zip_export and (
282 zip_compression is not None or zip_compression_level is not None
283 ):
284 raise CommandError(
285 "--zip-compression and --zip-compression-level require --zip",
286 )
288 compression_method = zip_compression or "deflated"
289 if self.zip_export:
290 if not compression_available(compression_method):
291 if compression_method == "zstd" and ZSTD is None:
292 raise CommandError(
293 "zstd compression requires Python 3.14 or newer",
294 )
295 raise CommandError(
296 f"Compression method '{compression_method}' is not "
297 f"available on this Python runtime",
298 )
299 level_msg = level_error(compression_method, zip_compression_level)
300 if level_msg is not None:
301 raise CommandError(level_msg)
303 sink: ExportSink
304 if self.zip_export:
305 sink = ZipExportSink(
306 self.target,
307 options["zip_name"],
308 delete=self.delete,
309 compression=COMPRESSION_METHODS[compression_method],
310 compresslevel=zip_compression_level,
311 )
312 else:
313 sink = DirectoryExportSink(
314 self.target,
315 compare_checksums=self.compare_checksums,
316 compare_json=self.compare_json,
317 delete=self.delete,
318 )
320 # Prevent any ongoing changes in the documents while exporting
321 with FileLock(settings.MEDIA_LOCK), sink:
322 self.dump(sink)
324 def dump(self, sink: ExportSink) -> None:
325 # 1. Create manifest, containing all correspondents, types, tags, storage
326 # paths, note, documents and ui_settings
327 _excluded_usernames = ["consumer", "AnonymousUser"]
328 manifest_key_to_object_query: dict[str, QuerySet[Any]] = {
329 "correspondents": Correspondent.objects.all(),
330 "tags": Tag.objects.all(),
331 "document_types": DocumentType.objects.all(),
332 "storage_paths": StoragePath.objects.all(),
333 "mail_accounts": MailAccount.objects.all(),
334 "mail_rules": MailRule.objects.all(),
335 "saved_views": SavedView.objects.all(),
336 "saved_view_filter_rules": SavedViewFilterRule.objects.all(),
337 "groups": Group.objects.all(),
338 "users": User.objects.exclude(
339 username__in=_excluded_usernames,
340 ).all(),
341 "ui_settings": UiSettings.objects.exclude(
342 user__username__in=_excluded_usernames,
343 ),
344 "content_types": ContentType.objects.all(),
345 "permissions": Permission.objects.all(),
346 "user_object_permissions": UserObjectPermission.objects.exclude(
347 user__username__in=_excluded_usernames,
348 ),
349 "group_object_permissions": GroupObjectPermission.objects.all(),
350 "workflow_triggers": WorkflowTrigger.objects.all(),
351 "workflow_actions": WorkflowAction.objects.all(),
352 "workflow_email_actions": WorkflowActionEmail.objects.all(),
353 "workflow_webhook_actions": WorkflowActionWebhook.objects.all(),
354 "workflows": Workflow.objects.all(),
355 "custom_fields": CustomField.objects.all(),
356 "custom_field_instances": CustomFieldInstance.global_objects.all(),
357 "document_barcodes": DocumentBarcode.objects.all(),
358 "app_configs": ApplicationConfiguration.objects.all(),
359 "notes": Note.global_objects.all(),
360 "documents": Document.global_objects.order_by("id").all(),
361 "share_links": ShareLink.global_objects.all(),
362 "share_link_bundles": ShareLinkBundle.objects.order_by("id").all(),
363 "social_accounts": SocialAccount.objects.exclude(
364 user__username__in=_excluded_usernames,
365 ),
366 "social_apps": SocialApp.objects.all(),
367 "social_tokens": SocialToken.objects.exclude(
368 account__user__username__in=_excluded_usernames,
369 ),
370 "authenticators": Authenticator.objects.exclude(
371 user__username__in=_excluded_usernames,
372 ),
373 }
375 if settings.AUDIT_LOG_ENABLED:
376 manifest_key_to_object_query["log_entries"] = LogEntry.objects.all()
378 # Crypto setup before streaming begins
379 if self.passphrase:
380 self.setup_crypto(passphrase=self.passphrase)
381 elif MailAccount.objects.count() > 0 or SocialToken.objects.count() > 0:
382 self.stdout.write(
383 self.style.NOTICE(
384 "No passphrase was given, sensitive fields will be in plaintext",
385 ),
386 )
388 document_manifest: list[dict] = []
389 share_link_bundle_manifest: list[dict] = []
391 with sink.stream("manifest.json") as handle:
392 writer = StreamingManifestWriter(handle)
393 with transaction.atomic():
394 for key, qs in manifest_key_to_object_query.items():
395 if key == "documents":
396 # Accumulate for file-copy loop; written to manifest after
397 for batch in serialize_queryset_batched(
398 qs,
399 batch_size=self.batch_size,
400 ):
401 for record in batch:
402 self._encrypt_record_inline(record)
403 document_manifest.extend(batch)
404 elif key == "share_link_bundles":
405 # Accumulate for file-copy loop; written to manifest after
406 for batch in serialize_queryset_batched(
407 qs,
408 batch_size=self.batch_size,
409 ):
410 for record in batch:
411 self._encrypt_record_inline(record)
412 share_link_bundle_manifest.extend(batch)
413 elif self.split_manifest and key in (
414 "notes",
415 "custom_field_instances",
416 "document_barcodes",
417 ):
418 # Written per-document in _write_split_manifest
419 pass
420 else:
421 for batch in serialize_queryset_batched(
422 qs,
423 batch_size=self.batch_size,
424 ):
425 for record in batch:
426 self._encrypt_record_inline(record)
427 writer.write_batch(batch)
429 share_link_bundle_map: dict[int, ShareLinkBundle] = {
430 b.pk: b
431 for b in ShareLinkBundle.objects.order_by("id").prefetch_related(
432 "documents",
433 )
434 }
436 # 2. Export files from each document
437 # document_manifest and this stream are both ordered by id from the
438 # same underlying rows, so zip them in lockstep instead of building
439 # a dict of every Document instance up front (QuerySetStream keeps
440 # only one batch of documents resident at a time).
441 documents_stream = QuerySetStream(
442 Document.global_objects.order_by("id"),
443 chunk_size=self.batch_size,
444 )
445 for document_dict, document in self.track(
446 zip(document_manifest, documents_stream, strict=True),
447 description="Exporting documents...",
448 total=len(document_manifest),
449 ):
450 # Both document_manifest and documents_stream come from the same
451 # Document.global_objects.order_by("id") query, taken while
452 # MEDIA_LOCK is held, so this should be unreachable -- it guards
453 # against silent data corruption if that invariant ever breaks.
454 if document.pk != document_dict["pk"]: # pragma: no cover
455 raise CommandError(
456 "Document export ordering mismatch: expected "
457 f"pk={document_dict['pk']}, got pk={document.pk}. "
458 "Documents may have changed during export.",
459 )
461 # generate a unique filename, then the arcnames for its files
462 base_name = self.generate_base_name(document)
463 original_arc, thumbnail_arc, archive_arc = (
464 self.generate_document_targets(document, base_name, document_dict)
465 )
467 if not self.data_only:
468 self.copy_document_files(
469 document,
470 sink,
471 original_arc,
472 thumbnail_arc,
473 archive_arc,
474 )
476 if self.split_manifest:
477 self._write_split_manifest(sink, document_dict, document, base_name)
478 else:
479 writer.write_record(document_dict)
481 for bundle_dict in share_link_bundle_manifest:
482 bundle = share_link_bundle_map[bundle_dict["pk"]]
483 bundle_arc = self.generate_share_link_bundle_target(
484 bundle,
485 bundle_dict,
486 )
487 if not self.data_only and bundle_arc is not None:
488 self.copy_share_link_bundle_file(bundle, sink, bundle_arc)
489 writer.write_record(bundle_dict)
491 writer.close()
493 # 3. Write version (and crypto params) to metadata.json
494 # Django stores most crypto values in the field itself; we store
495 # them once here for the whole export
496 metadata: dict[str, str | int | dict[str, str | int]] = {
497 "version": version.__full_version_str__,
498 }
499 if self.passphrase:
500 metadata.update(self.get_crypt_params())
501 sink.add_json(metadata, "metadata.json")
503 def generate_base_name(self, document: Document) -> Path:
504 """
505 Generates a unique name for the document, one which hasn't already been exported (or will be)
506 """
507 filename_counter = 0
508 while True:
509 if self.use_filename_format:
510 base_name = generate_filename(
511 document,
512 counter=filename_counter,
513 )
514 else:
515 base_name = document.get_public_filename(counter=filename_counter)
517 if base_name not in self.exported_files:
518 self.exported_files.add(base_name)
519 break
520 else:
521 filename_counter += 1
522 return Path(base_name)
524 def generate_document_targets(
525 self,
526 document: Document,
527 base_name: Path,
528 document_dict: dict,
529 ) -> tuple[str, str | None, str | None]:
530 """
531 Generates the relative POSIX arcnames for a document's original, thumbnail
532 and archive files (depending on settings), and records them in the manifest.
533 """
534 original_name = base_name
535 if self.use_folder_prefix:
536 original_name = Path("originals") / original_name
537 original_arc = original_name.as_posix()
538 document_dict[EXPORTER_FILE_NAME] = original_arc
540 if not self.no_thumbnail:
541 thumbnail_name = base_name.parent / (base_name.stem + "-thumbnail.webp")
542 if self.use_folder_prefix:
543 thumbnail_name = Path("thumbnails") / thumbnail_name
544 thumbnail_arc = thumbnail_name.as_posix()
545 document_dict[EXPORTER_THUMBNAIL_NAME] = thumbnail_arc
546 else:
547 thumbnail_arc = None
549 if not self.no_archive and document.has_archive_version:
550 archive_name = base_name.parent / (base_name.stem + "-archive.pdf")
551 if self.use_folder_prefix:
552 archive_name = Path("archive") / archive_name
553 archive_arc = archive_name.as_posix()
554 document_dict[EXPORTER_ARCHIVE_NAME] = archive_arc
555 else:
556 archive_arc = None
558 return original_arc, thumbnail_arc, archive_arc
560 def copy_document_files(
561 self,
562 document: Document,
563 sink: ExportSink,
564 original_arc: str,
565 thumbnail_arc: str | None,
566 archive_arc: str | None,
567 ) -> None:
568 """
569 Hands the document's files to the sink (original, thumbnail, archive).
570 """
571 sink.add_file(document.source_path, original_arc, checksum=document.checksum)
573 if thumbnail_arc:
574 sink.add_file(document.thumbnail_path, thumbnail_arc)
576 if archive_arc:
577 if TYPE_CHECKING:
578 assert isinstance(document.archive_path, Path)
579 sink.add_file(
580 document.archive_path,
581 archive_arc,
582 checksum=document.archive_checksum,
583 )
585 def generate_share_link_bundle_target(
586 self,
587 bundle: ShareLinkBundle,
588 bundle_dict: dict,
589 ) -> str | None:
590 """
591 Generates the relative POSIX arcname for a share link bundle file, if any.
592 """
593 if not bundle.file_path:
594 return None
596 stored_bundle_path = Path(bundle.file_path)
597 portable_bundle_path = (
598 stored_bundle_path
599 if not stored_bundle_path.is_absolute()
600 else Path(stored_bundle_path.name)
601 )
602 export_bundle_path = Path("share_link_bundles") / portable_bundle_path
604 bundle_dict["fields"]["file_path"] = portable_bundle_path.as_posix()
605 bundle_dict[EXPORTER_SHARE_LINK_BUNDLE_NAME] = export_bundle_path.as_posix()
607 return export_bundle_path.as_posix()
609 def copy_share_link_bundle_file(
610 self,
611 bundle: ShareLinkBundle,
612 sink: ExportSink,
613 bundle_arc: str,
614 ) -> None:
615 """
616 Hands a share link bundle ZIP to the sink.
617 """
618 bundle_source_path = bundle.absolute_file_path
619 if bundle_source_path is None:
620 raise FileNotFoundError(f"Share link bundle {bundle.pk} has no file path")
622 sink.add_file(bundle_source_path, bundle_arc)
624 def _encrypt_record_inline(self, record: dict) -> None:
625 """Encrypt sensitive fields in a single record, if passphrase is set."""
626 if not self.passphrase:
627 return
628 fields = self.CRYPT_FIELDS_BY_MODEL.get(record.get("model", ""))
629 if fields:
630 for field in fields:
631 if record["fields"].get(field):
632 record["fields"][field] = self.encrypt_string(
633 value=record["fields"][field],
634 )
636 def _write_split_manifest(
637 self,
638 sink: ExportSink,
639 document_dict: dict,
640 document: Document,
641 base_name: Path,
642 ) -> None:
643 """Write per-document manifest file for --split-manifest mode."""
644 content = [document_dict]
645 content.extend(
646 serializers.serialize(
647 "python",
648 Note.global_objects.filter(document=document),
649 ),
650 )
651 content.extend(
652 serializers.serialize(
653 "python",
654 CustomFieldInstance.global_objects.filter(document=document),
655 ),
656 )
657 content.extend(
658 serializers.serialize(
659 "python",
660 DocumentBarcode.objects.filter(document=document),
661 ),
662 )
663 manifest_name = base_name.with_name(f"{base_name.stem}-manifest.json")
664 if self.use_folder_prefix:
665 manifest_name = Path("json") / manifest_name
666 sink.add_json(content, manifest_name.as_posix())