Coverage for documents/management/commands/document_exporter.py: 0%

258 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1import os 

2from itertools import islice 

3from pathlib import Path 

4from typing import TYPE_CHECKING 

5from typing import Any 

6 

7from allauth.mfa.models import Authenticator 

8from allauth.socialaccount.models import SocialAccount 

9from allauth.socialaccount.models import SocialApp 

10from allauth.socialaccount.models import SocialToken 

11from django.conf import settings 

12from django.contrib.auth.models import Group 

13from django.contrib.auth.models import Permission 

14from django.contrib.auth.models import User 

15from django.contrib.contenttypes.models import ContentType 

16from django.core import serializers 

17from django.core.management.base import CommandError 

18from django.db import transaction 

19from django.utils import timezone 

20from filelock import FileLock 

21from guardian.models import GroupObjectPermission 

22from guardian.models import UserObjectPermission 

23 

24if TYPE_CHECKING: 

25 from collections.abc import Generator 

26 

27 from django.db.models import QuerySet 

28 

29if settings.AUDIT_LOG_ENABLED: 

30 from auditlog.models import LogEntry 

31 

32from documents.export.compression import COMPRESSION_CHOICES 

33from documents.export.compression import COMPRESSION_METHODS 

34from documents.export.compression import ZSTD 

35from documents.export.compression import compression_available 

36from documents.export.compression import level_error 

37from documents.export.sinks import DirectoryExportSink 

38from documents.export.sinks import ExportSink 

39from documents.export.sinks import StreamingManifestWriter 

40from documents.export.sinks import ZipExportSink 

41from documents.file_handling import generate_filename 

42from documents.management.commands.base import PaperlessCommand 

43from documents.management.commands.mixins import CryptMixin 

44from documents.models import Correspondent 

45from documents.models import CustomField 

46from documents.models import CustomFieldInstance 

47from documents.models import Document 

48from documents.models import DocumentBarcode 

49from documents.models import DocumentType 

50from documents.models import Note 

51from documents.models import SavedView 

52from documents.models import SavedViewFilterRule 

53from documents.models import ShareLink 

54from documents.models import ShareLinkBundle 

55from documents.models import StoragePath 

56from documents.models import Tag 

57from documents.models import UiSettings 

58from documents.models import Workflow 

59from documents.models import WorkflowAction 

60from documents.models import WorkflowActionEmail 

61from documents.models import WorkflowActionWebhook 

62from documents.models import WorkflowTrigger 

63from documents.settings import EXPORTER_ARCHIVE_NAME 

64from documents.settings import EXPORTER_FILE_NAME 

65from documents.settings import EXPORTER_SHARE_LINK_BUNDLE_NAME 

66from documents.settings import EXPORTER_THUMBNAIL_NAME 

67from documents.utils import QuerySetStream 

68from paperless import version 

69from paperless.models import ApplicationConfiguration 

70from paperless_mail.models import MailAccount 

71from paperless_mail.models import MailRule 

72 

73 

74def serialize_queryset_batched( 

75 queryset: "QuerySet[Any]", 

76 *, 

77 batch_size: int = 500, 

78) -> "Generator[list[dict], None, None]": 

79 """Yield batches of serialized records from a QuerySet. 

80 

81 Each batch is a list of dicts in Django's Python serialization format. 

82 Uses QuerySet.iterator() to avoid loading the full queryset into memory, 

83 and islice to collect chunk-sized batches serialized in a single call. 

84 """ 

85 iterator = queryset.iterator(chunk_size=batch_size) 

86 while chunk := list(islice(iterator, batch_size)): 

87 yield serializers.serialize("python", chunk) 

88 

89 

90class Command(CryptMixin, PaperlessCommand): 

91 help = ( 

92 "Decrypt and rename all files in our collection into a given target " 

93 "directory. And include a manifest file containing document data for " 

94 "easy import." 

95 ) 

96 

97 supports_progress_bar = True 

98 supports_multiprocessing = False 

99 

100 def add_arguments(self, parser) -> None: 

101 super().add_arguments(parser) 

102 parser.add_argument("target") 

103 

104 parser.add_argument( 

105 "-c", 

106 "--compare-checksums", 

107 default=False, 

108 action="store_true", 

109 help=( 

110 "Compare file checksums when determining whether to export " 

111 "a file or not. If not specified, file size and time " 

112 "modified is used instead." 

113 ), 

114 ) 

115 

116 parser.add_argument( 

117 "-cj", 

118 "--compare-json", 

119 default=False, 

120 action="store_true", 

121 help=( 

122 "Compare json file checksums when determining whether to " 

123 "export a json file or not (manifest or metadata). " 

124 "If not specified, the file is always exported." 

125 ), 

126 ) 

127 

128 parser.add_argument( 

129 "-d", 

130 "--delete", 

131 default=False, 

132 action="store_true", 

133 help=( 

134 "After exporting, delete files in the export directory that " 

135 "do not belong to the current export, such as files from " 

136 "deleted documents." 

137 ), 

138 ) 

139 

140 parser.add_argument( 

141 "-f", 

142 "--use-filename-format", 

143 default=False, 

144 action="store_true", 

145 help=( 

146 "Use PAPERLESS_FILENAME_FORMAT for storing files in the " 

147 "export directory, if configured." 

148 ), 

149 ) 

150 

151 parser.add_argument( 

152 "-na", 

153 "--no-archive", 

154 default=False, 

155 action="store_true", 

156 help="Avoid exporting archive files", 

157 ) 

158 

159 parser.add_argument( 

160 "-nt", 

161 "--no-thumbnail", 

162 default=False, 

163 action="store_true", 

164 help="Avoid exporting thumbnail files", 

165 ) 

166 

167 parser.add_argument( 

168 "-p", 

169 "--use-folder-prefix", 

170 default=False, 

171 action="store_true", 

172 help=( 

173 "Export files in dedicated folders according to their nature: " 

174 "archive, originals or thumbnails" 

175 ), 

176 ) 

177 

178 parser.add_argument( 

179 "-sm", 

180 "--split-manifest", 

181 default=False, 

182 action="store_true", 

183 help="Export document information in individual manifest json files.", 

184 ) 

185 

186 parser.add_argument( 

187 "-z", 

188 "--zip", 

189 default=False, 

190 action="store_true", 

191 help="Export the documents to a zip file in the given directory", 

192 ) 

193 

194 parser.add_argument( 

195 "-zn", 

196 "--zip-name", 

197 default=f"export-{timezone.localdate().isoformat()}", 

198 help="Sets the export zip file name", 

199 ) 

200 

201 parser.add_argument( 

202 "--zip-compression", 

203 choices=COMPRESSION_CHOICES, 

204 default=None, 

205 help=( 

206 "Compression method for the export zip (requires --zip). " 

207 "Default: deflated. 'zstd' requires Python 3.14+ on both the " 

208 "exporting and importing machine." 

209 ), 

210 ) 

211 

212 parser.add_argument( 

213 "--zip-compression-level", 

214 type=int, 

215 default=None, 

216 help=( 

217 "Compression level for the export zip (requires --zip). " 

218 "deflated: 0-9, bzip2: 1-9, zstd: -22..22; ignored for " 

219 "stored/lzma." 

220 ), 

221 ) 

222 

223 parser.add_argument( 

224 "--data-only", 

225 default=False, 

226 action="store_true", 

227 help="If set, only the database will be imported, not files", 

228 ) 

229 

230 parser.add_argument( 

231 "--passphrase", 

232 help="If provided, is used to encrypt sensitive data in the export", 

233 ) 

234 

235 parser.add_argument( 

236 "--batch-size", 

237 type=int, 

238 default=500, 

239 help=( 

240 "Number of records to process per batch during serialization. " 

241 "Lower values reduce peak memory usage; higher values improve " 

242 "throughput. Default: 500." 

243 ), 

244 ) 

245 

246 def handle(self, *args, **options) -> None: 

247 self.target = Path(options["target"]).resolve() 

248 self.split_manifest: bool = options["split_manifest"] 

249 self.compare_checksums: bool = options["compare_checksums"] 

250 self.compare_json: bool = options["compare_json"] 

251 self.use_filename_format: bool = options["use_filename_format"] 

252 self.use_folder_prefix: bool = options["use_folder_prefix"] 

253 self.delete: bool = options["delete"] 

254 self.no_archive: bool = options["no_archive"] 

255 self.no_thumbnail: bool = options["no_thumbnail"] 

256 self.zip_export: bool = options["zip"] 

257 self.data_only: bool = options["data_only"] 

258 self.passphrase: str | None = options.get("passphrase") 

259 self.batch_size: int = options["batch_size"] 

260 

261 self.exported_files: set[str] = set() 

262 

263 if self.zip_export and (self.compare_checksums or self.compare_json): 

264 raise CommandError( 

265 "--compare-checksums and --compare-json have no effect when " 

266 "used with --zip", 

267 ) 

268 

269 if not self.target.exists(): 

270 raise CommandError("That path doesn't exist") 

271 

272 if not self.target.is_dir(): 

273 raise CommandError("That path isn't a directory") 

274 

275 if not os.access(self.target, os.W_OK): 

276 raise CommandError("That path doesn't appear to be writable") 

277 

278 zip_compression: str | None = options["zip_compression"] 

279 zip_compression_level: int | None = options["zip_compression_level"] 

280 

281 if not self.zip_export and ( 

282 zip_compression is not None or zip_compression_level is not None 

283 ): 

284 raise CommandError( 

285 "--zip-compression and --zip-compression-level require --zip", 

286 ) 

287 

288 compression_method = zip_compression or "deflated" 

289 if self.zip_export: 

290 if not compression_available(compression_method): 

291 if compression_method == "zstd" and ZSTD is None: 

292 raise CommandError( 

293 "zstd compression requires Python 3.14 or newer", 

294 ) 

295 raise CommandError( 

296 f"Compression method '{compression_method}' is not " 

297 f"available on this Python runtime", 

298 ) 

299 level_msg = level_error(compression_method, zip_compression_level) 

300 if level_msg is not None: 

301 raise CommandError(level_msg) 

302 

303 sink: ExportSink 

304 if self.zip_export: 

305 sink = ZipExportSink( 

306 self.target, 

307 options["zip_name"], 

308 delete=self.delete, 

309 compression=COMPRESSION_METHODS[compression_method], 

310 compresslevel=zip_compression_level, 

311 ) 

312 else: 

313 sink = DirectoryExportSink( 

314 self.target, 

315 compare_checksums=self.compare_checksums, 

316 compare_json=self.compare_json, 

317 delete=self.delete, 

318 ) 

319 

320 # Prevent any ongoing changes in the documents while exporting 

321 with FileLock(settings.MEDIA_LOCK), sink: 

322 self.dump(sink) 

323 

324 def dump(self, sink: ExportSink) -> None: 

325 # 1. Create manifest, containing all correspondents, types, tags, storage 

326 # paths, note, documents and ui_settings 

327 _excluded_usernames = ["consumer", "AnonymousUser"] 

328 manifest_key_to_object_query: dict[str, QuerySet[Any]] = { 

329 "correspondents": Correspondent.objects.all(), 

330 "tags": Tag.objects.all(), 

331 "document_types": DocumentType.objects.all(), 

332 "storage_paths": StoragePath.objects.all(), 

333 "mail_accounts": MailAccount.objects.all(), 

334 "mail_rules": MailRule.objects.all(), 

335 "saved_views": SavedView.objects.all(), 

336 "saved_view_filter_rules": SavedViewFilterRule.objects.all(), 

337 "groups": Group.objects.all(), 

338 "users": User.objects.exclude( 

339 username__in=_excluded_usernames, 

340 ).all(), 

341 "ui_settings": UiSettings.objects.exclude( 

342 user__username__in=_excluded_usernames, 

343 ), 

344 "content_types": ContentType.objects.all(), 

345 "permissions": Permission.objects.all(), 

346 "user_object_permissions": UserObjectPermission.objects.exclude( 

347 user__username__in=_excluded_usernames, 

348 ), 

349 "group_object_permissions": GroupObjectPermission.objects.all(), 

350 "workflow_triggers": WorkflowTrigger.objects.all(), 

351 "workflow_actions": WorkflowAction.objects.all(), 

352 "workflow_email_actions": WorkflowActionEmail.objects.all(), 

353 "workflow_webhook_actions": WorkflowActionWebhook.objects.all(), 

354 "workflows": Workflow.objects.all(), 

355 "custom_fields": CustomField.objects.all(), 

356 "custom_field_instances": CustomFieldInstance.global_objects.all(), 

357 "document_barcodes": DocumentBarcode.objects.all(), 

358 "app_configs": ApplicationConfiguration.objects.all(), 

359 "notes": Note.global_objects.all(), 

360 "documents": Document.global_objects.order_by("id").all(), 

361 "share_links": ShareLink.global_objects.all(), 

362 "share_link_bundles": ShareLinkBundle.objects.order_by("id").all(), 

363 "social_accounts": SocialAccount.objects.exclude( 

364 user__username__in=_excluded_usernames, 

365 ), 

366 "social_apps": SocialApp.objects.all(), 

367 "social_tokens": SocialToken.objects.exclude( 

368 account__user__username__in=_excluded_usernames, 

369 ), 

370 "authenticators": Authenticator.objects.exclude( 

371 user__username__in=_excluded_usernames, 

372 ), 

373 } 

374 

375 if settings.AUDIT_LOG_ENABLED: 

376 manifest_key_to_object_query["log_entries"] = LogEntry.objects.all() 

377 

378 # Crypto setup before streaming begins 

379 if self.passphrase: 

380 self.setup_crypto(passphrase=self.passphrase) 

381 elif MailAccount.objects.count() > 0 or SocialToken.objects.count() > 0: 

382 self.stdout.write( 

383 self.style.NOTICE( 

384 "No passphrase was given, sensitive fields will be in plaintext", 

385 ), 

386 ) 

387 

388 document_manifest: list[dict] = [] 

389 share_link_bundle_manifest: list[dict] = [] 

390 

391 with sink.stream("manifest.json") as handle: 

392 writer = StreamingManifestWriter(handle) 

393 with transaction.atomic(): 

394 for key, qs in manifest_key_to_object_query.items(): 

395 if key == "documents": 

396 # Accumulate for file-copy loop; written to manifest after 

397 for batch in serialize_queryset_batched( 

398 qs, 

399 batch_size=self.batch_size, 

400 ): 

401 for record in batch: 

402 self._encrypt_record_inline(record) 

403 document_manifest.extend(batch) 

404 elif key == "share_link_bundles": 

405 # Accumulate for file-copy loop; written to manifest after 

406 for batch in serialize_queryset_batched( 

407 qs, 

408 batch_size=self.batch_size, 

409 ): 

410 for record in batch: 

411 self._encrypt_record_inline(record) 

412 share_link_bundle_manifest.extend(batch) 

413 elif self.split_manifest and key in ( 

414 "notes", 

415 "custom_field_instances", 

416 "document_barcodes", 

417 ): 

418 # Written per-document in _write_split_manifest 

419 pass 

420 else: 

421 for batch in serialize_queryset_batched( 

422 qs, 

423 batch_size=self.batch_size, 

424 ): 

425 for record in batch: 

426 self._encrypt_record_inline(record) 

427 writer.write_batch(batch) 

428 

429 share_link_bundle_map: dict[int, ShareLinkBundle] = { 

430 b.pk: b 

431 for b in ShareLinkBundle.objects.order_by("id").prefetch_related( 

432 "documents", 

433 ) 

434 } 

435 

436 # 2. Export files from each document 

437 # document_manifest and this stream are both ordered by id from the 

438 # same underlying rows, so zip them in lockstep instead of building 

439 # a dict of every Document instance up front (QuerySetStream keeps 

440 # only one batch of documents resident at a time). 

441 documents_stream = QuerySetStream( 

442 Document.global_objects.order_by("id"), 

443 chunk_size=self.batch_size, 

444 ) 

445 for document_dict, document in self.track( 

446 zip(document_manifest, documents_stream, strict=True), 

447 description="Exporting documents...", 

448 total=len(document_manifest), 

449 ): 

450 # Both document_manifest and documents_stream come from the same 

451 # Document.global_objects.order_by("id") query, taken while 

452 # MEDIA_LOCK is held, so this should be unreachable -- it guards 

453 # against silent data corruption if that invariant ever breaks. 

454 if document.pk != document_dict["pk"]: # pragma: no cover 

455 raise CommandError( 

456 "Document export ordering mismatch: expected " 

457 f"pk={document_dict['pk']}, got pk={document.pk}. " 

458 "Documents may have changed during export.", 

459 ) 

460 

461 # generate a unique filename, then the arcnames for its files 

462 base_name = self.generate_base_name(document) 

463 original_arc, thumbnail_arc, archive_arc = ( 

464 self.generate_document_targets(document, base_name, document_dict) 

465 ) 

466 

467 if not self.data_only: 

468 self.copy_document_files( 

469 document, 

470 sink, 

471 original_arc, 

472 thumbnail_arc, 

473 archive_arc, 

474 ) 

475 

476 if self.split_manifest: 

477 self._write_split_manifest(sink, document_dict, document, base_name) 

478 else: 

479 writer.write_record(document_dict) 

480 

481 for bundle_dict in share_link_bundle_manifest: 

482 bundle = share_link_bundle_map[bundle_dict["pk"]] 

483 bundle_arc = self.generate_share_link_bundle_target( 

484 bundle, 

485 bundle_dict, 

486 ) 

487 if not self.data_only and bundle_arc is not None: 

488 self.copy_share_link_bundle_file(bundle, sink, bundle_arc) 

489 writer.write_record(bundle_dict) 

490 

491 writer.close() 

492 

493 # 3. Write version (and crypto params) to metadata.json 

494 # Django stores most crypto values in the field itself; we store 

495 # them once here for the whole export 

496 metadata: dict[str, str | int | dict[str, str | int]] = { 

497 "version": version.__full_version_str__, 

498 } 

499 if self.passphrase: 

500 metadata.update(self.get_crypt_params()) 

501 sink.add_json(metadata, "metadata.json") 

502 

503 def generate_base_name(self, document: Document) -> Path: 

504 """ 

505 Generates a unique name for the document, one which hasn't already been exported (or will be) 

506 """ 

507 filename_counter = 0 

508 while True: 

509 if self.use_filename_format: 

510 base_name = generate_filename( 

511 document, 

512 counter=filename_counter, 

513 ) 

514 else: 

515 base_name = document.get_public_filename(counter=filename_counter) 

516 

517 if base_name not in self.exported_files: 

518 self.exported_files.add(base_name) 

519 break 

520 else: 

521 filename_counter += 1 

522 return Path(base_name) 

523 

524 def generate_document_targets( 

525 self, 

526 document: Document, 

527 base_name: Path, 

528 document_dict: dict, 

529 ) -> tuple[str, str | None, str | None]: 

530 """ 

531 Generates the relative POSIX arcnames for a document's original, thumbnail 

532 and archive files (depending on settings), and records them in the manifest. 

533 """ 

534 original_name = base_name 

535 if self.use_folder_prefix: 

536 original_name = Path("originals") / original_name 

537 original_arc = original_name.as_posix() 

538 document_dict[EXPORTER_FILE_NAME] = original_arc 

539 

540 if not self.no_thumbnail: 

541 thumbnail_name = base_name.parent / (base_name.stem + "-thumbnail.webp") 

542 if self.use_folder_prefix: 

543 thumbnail_name = Path("thumbnails") / thumbnail_name 

544 thumbnail_arc = thumbnail_name.as_posix() 

545 document_dict[EXPORTER_THUMBNAIL_NAME] = thumbnail_arc 

546 else: 

547 thumbnail_arc = None 

548 

549 if not self.no_archive and document.has_archive_version: 

550 archive_name = base_name.parent / (base_name.stem + "-archive.pdf") 

551 if self.use_folder_prefix: 

552 archive_name = Path("archive") / archive_name 

553 archive_arc = archive_name.as_posix() 

554 document_dict[EXPORTER_ARCHIVE_NAME] = archive_arc 

555 else: 

556 archive_arc = None 

557 

558 return original_arc, thumbnail_arc, archive_arc 

559 

560 def copy_document_files( 

561 self, 

562 document: Document, 

563 sink: ExportSink, 

564 original_arc: str, 

565 thumbnail_arc: str | None, 

566 archive_arc: str | None, 

567 ) -> None: 

568 """ 

569 Hands the document's files to the sink (original, thumbnail, archive). 

570 """ 

571 sink.add_file(document.source_path, original_arc, checksum=document.checksum) 

572 

573 if thumbnail_arc: 

574 sink.add_file(document.thumbnail_path, thumbnail_arc) 

575 

576 if archive_arc: 

577 if TYPE_CHECKING: 

578 assert isinstance(document.archive_path, Path) 

579 sink.add_file( 

580 document.archive_path, 

581 archive_arc, 

582 checksum=document.archive_checksum, 

583 ) 

584 

585 def generate_share_link_bundle_target( 

586 self, 

587 bundle: ShareLinkBundle, 

588 bundle_dict: dict, 

589 ) -> str | None: 

590 """ 

591 Generates the relative POSIX arcname for a share link bundle file, if any. 

592 """ 

593 if not bundle.file_path: 

594 return None 

595 

596 stored_bundle_path = Path(bundle.file_path) 

597 portable_bundle_path = ( 

598 stored_bundle_path 

599 if not stored_bundle_path.is_absolute() 

600 else Path(stored_bundle_path.name) 

601 ) 

602 export_bundle_path = Path("share_link_bundles") / portable_bundle_path 

603 

604 bundle_dict["fields"]["file_path"] = portable_bundle_path.as_posix() 

605 bundle_dict[EXPORTER_SHARE_LINK_BUNDLE_NAME] = export_bundle_path.as_posix() 

606 

607 return export_bundle_path.as_posix() 

608 

609 def copy_share_link_bundle_file( 

610 self, 

611 bundle: ShareLinkBundle, 

612 sink: ExportSink, 

613 bundle_arc: str, 

614 ) -> None: 

615 """ 

616 Hands a share link bundle ZIP to the sink. 

617 """ 

618 bundle_source_path = bundle.absolute_file_path 

619 if bundle_source_path is None: 

620 raise FileNotFoundError(f"Share link bundle {bundle.pk} has no file path") 

621 

622 sink.add_file(bundle_source_path, bundle_arc) 

623 

624 def _encrypt_record_inline(self, record: dict) -> None: 

625 """Encrypt sensitive fields in a single record, if passphrase is set.""" 

626 if not self.passphrase: 

627 return 

628 fields = self.CRYPT_FIELDS_BY_MODEL.get(record.get("model", "")) 

629 if fields: 

630 for field in fields: 

631 if record["fields"].get(field): 

632 record["fields"][field] = self.encrypt_string( 

633 value=record["fields"][field], 

634 ) 

635 

636 def _write_split_manifest( 

637 self, 

638 sink: ExportSink, 

639 document_dict: dict, 

640 document: Document, 

641 base_name: Path, 

642 ) -> None: 

643 """Write per-document manifest file for --split-manifest mode.""" 

644 content = [document_dict] 

645 content.extend( 

646 serializers.serialize( 

647 "python", 

648 Note.global_objects.filter(document=document), 

649 ), 

650 ) 

651 content.extend( 

652 serializers.serialize( 

653 "python", 

654 CustomFieldInstance.global_objects.filter(document=document), 

655 ), 

656 ) 

657 content.extend( 

658 serializers.serialize( 

659 "python", 

660 DocumentBarcode.objects.filter(document=document), 

661 ), 

662 ) 

663 manifest_name = base_name.with_name(f"{base_name.stem}-manifest.json") 

664 if self.use_folder_prefix: 

665 manifest_name = Path("json") / manifest_name 

666 sink.add_json(content, manifest_name.as_posix())