Coverage for documents/management/commands/document_importer.py: 0%

372 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1import json 

2import logging 

3import os 

4import tempfile 

5from collections import defaultdict 

6from collections.abc import Generator 

7from contextlib import contextmanager 

8from pathlib import Path 

9from typing import TypeAlias 

10from zipfile import ZipFile 

11from zipfile import is_zipfile 

12 

13import ijson 

14from django.apps import apps 

15from django.conf import settings 

16from django.contrib.auth.models import Permission 

17from django.contrib.auth.models import User 

18from django.contrib.contenttypes.models import ContentType 

19from django.core.exceptions import FieldDoesNotExist 

20from django.core.management import call_command 

21from django.core.management.base import CommandError 

22from django.core.management.color import no_style 

23from django.core.serializers.base import DeserializationError 

24from django.db import IntegrityError 

25from django.db import connection 

26from django.db import models as django_models 

27from django.db import transaction 

28from django.db.models import GeneratedField 

29from django.db.models import Model 

30from django.db.models.signals import m2m_changed 

31from django.db.models.signals import post_save 

32from filelock import FileLock 

33from guardian.shortcuts import clear_ct_cache 

34 

35from documents.export.compression import compress_type_readable 

36from documents.export.compression import unreadable_method_names 

37from documents.file_handling import create_source_path_directory 

38from documents.management.commands.base import PaperlessCommand 

39from documents.management.commands.mixins import CryptMixin 

40from documents.models import Correspondent 

41from documents.models import CustomField 

42from documents.models import CustomFieldInstance 

43from documents.models import Document 

44from documents.models import DocumentType 

45from documents.models import Note 

46from documents.models import ShareLinkBundle 

47from documents.models import Tag 

48from documents.settings import EXPORTER_ARCHIVE_NAME 

49from documents.settings import EXPORTER_CRYPTO_SETTINGS_NAME 

50from documents.settings import EXPORTER_FILE_NAME 

51from documents.settings import EXPORTER_SHARE_LINK_BUNDLE_NAME 

52from documents.settings import EXPORTER_THUMBNAIL_NAME 

53from documents.signals.handlers import check_paths_and_prune_custom_fields 

54from documents.signals.handlers import update_filename_and_move_files 

55from documents.utils import copy_file_with_basic_stats 

56from paperless import version 

57 

58if settings.AUDIT_LOG_ENABLED: 

59 from auditlog.registry import auditlog 

60 

61# Maps M2M field names to the list of related PKs to apply after bulk_create. 

62M2MData: TypeAlias = dict[str, list[int]] 

63 

64 

65def iter_manifest_records(path: Path) -> Generator[dict, None, None]: 

66 """Yield records one at a time from a manifest JSON array via ijson.""" 

67 try: 

68 with path.open("rb") as f: 

69 yield from ijson.items(f, "item") 

70 except ijson.JSONError as e: 

71 raise CommandError(f"Failed to parse manifest file {path}: {e}") from e 

72 

73 

74def _deserialize_record( 

75 record: dict, 

76) -> tuple[type[Model], Model, M2MData]: 

77 """ 

78 Convert a single manifest record dict into a model instance and M2M data. 

79 

80 Returns (Model class, unsaved instance, m2m_data) where m2m_data maps 

81 M2M field names to lists of integer PKs to be applied after the instance 

82 is saved via bulk_create. 

83 

84 Raises DeserializationError for unknown models or bad field values. 

85 Raises FieldDoesNotExist for fields not present on the model. 

86 

87 Note: CommandError from iter_manifest_records (malformed JSON mid-stream) 

88 propagates through the caller unchanged, it is not caught here. 

89 """ 

90 model_label = record["model"] 

91 pk_value = record.get("pk") 

92 

93 try: 

94 Model = apps.get_model(model_label) 

95 except (LookupError, TypeError) as e: 

96 raise DeserializationError( 

97 f"Invalid model identifier: {model_label}", 

98 ) from e 

99 

100 data: dict = {} 

101 m2m_data: M2MData = {} 

102 

103 try: 

104 data[Model._meta.pk.attname] = Model._meta.pk.to_python(pk_value) 

105 except Exception as e: 

106 raise DeserializationError( 

107 f"Could not coerce pk={pk_value} for {model_label}: {e}", 

108 ) from e 

109 

110 for field_name, field_value in record.get("fields", {}).items(): 

111 field = Model._meta.get_field(field_name) 

112 remote = field.remote_field 

113 

114 if isinstance(remote, django_models.ManyToManyRel): 

115 # Collect M2M PKs; .set() is called after bulk_create in flush_model. 

116 target_pk = field.related_model._meta.pk 

117 m2m_data[field.name] = [ 

118 target_pk.to_python(pk) for pk in (field_value or []) 

119 ] 

120 

121 elif isinstance(remote, django_models.ManyToOneRel): 

122 # FK: store the integer PK on field.attname (e.g. correspondent_id) 

123 # to avoid triggering the descriptor and avoid an extra DB lookup. 

124 if field_value is None: 

125 data[field.attname] = None 

126 else: 

127 data[field.attname] = field.related_model._meta.pk.to_python( 

128 field_value, 

129 ) 

130 

131 else: 

132 try: 

133 data[field.name] = field.to_python(field_value) 

134 except Exception as e: 

135 raise DeserializationError( 

136 f"Could not coerce {field_name}={field_value!r} " 

137 f"for {model_label}(pk={pk_value}): {e}", 

138 ) from e 

139 

140 return Model, Model(**data), m2m_data 

141 

142 

143def _iter_document_copy_records( 

144 manifest_paths: list[Path], 

145) -> Generator[dict, None, None]: 

146 """Yield one lightweight dict per Document record without buffering all records.""" 

147 for manifest_path in manifest_paths: 

148 for record in iter_manifest_records(manifest_path): 

149 if record["model"] == "documents.document": 

150 yield { 

151 "pk": record["pk"], 

152 EXPORTER_FILE_NAME: record[EXPORTER_FILE_NAME], 

153 EXPORTER_THUMBNAIL_NAME: record.get(EXPORTER_THUMBNAIL_NAME), 

154 EXPORTER_ARCHIVE_NAME: record.get(EXPORTER_ARCHIVE_NAME), 

155 } 

156 

157 

158def _iter_share_link_bundle_copy_records( 

159 manifest_paths: list[Path], 

160) -> Generator[dict, None, None]: 

161 """Yield one dict per ShareLinkBundle record that has a bundle file.""" 

162 for manifest_path in manifest_paths: 

163 for record in iter_manifest_records(manifest_path): 

164 if record["model"] == "documents.sharelinkbundle" and record.get( 

165 EXPORTER_SHARE_LINK_BUNDLE_NAME, 

166 ): 

167 yield { 

168 "pk": record["pk"], 

169 EXPORTER_SHARE_LINK_BUNDLE_NAME: record[ 

170 EXPORTER_SHARE_LINK_BUNDLE_NAME 

171 ], 

172 } 

173 

174 

175@contextmanager 

176def disable_signal(sig, receiver, sender, *, weak: bool | None = None) -> Generator: 

177 try: 

178 sig.disconnect(receiver=receiver, sender=sender) 

179 yield 

180 finally: 

181 kwargs = {"weak": weak} if weak is not None else {} 

182 sig.connect(receiver=receiver, sender=sender, **kwargs) 

183 

184 

185class Command(CryptMixin, PaperlessCommand): 

186 help = ( 

187 "Using a manifest.json file, load the data from there, and import the " 

188 "documents it refers to." 

189 ) 

190 

191 supports_progress_bar = True 

192 supports_multiprocessing = False 

193 

194 def add_arguments(self, parser) -> None: 

195 super().add_arguments(parser) 

196 parser.add_argument("source") 

197 

198 parser.add_argument( 

199 "--data-only", 

200 default=False, 

201 action="store_true", 

202 help="If set, only the database will be exported, not files", 

203 ) 

204 

205 parser.add_argument( 

206 "--passphrase", 

207 help="If provided, is used to sensitive fields in the export", 

208 ) 

209 

210 parser.add_argument( 

211 "--batch-size", 

212 type=int, 

213 default=500, 

214 help="Number of records to insert per batch during database load. " 

215 "Lower values reduce peak memory usage.", 

216 ) 

217 

218 def pre_check(self) -> None: 

219 """ 

220 Runs some initial checks against the state of the install and source, including: 

221 - Does the target exist? 

222 - Can we access the target? 

223 - Does the target have a manifest file? 

224 - Are there existing files in the document folders? 

225 - Are there existing users or documents in the database? 

226 """ 

227 

228 def pre_check_maybe_not_empty() -> None: 

229 # Skip this check if operating only on the database 

230 # We can expect data to exist in that case 

231 if not self.data_only: 

232 for document_dir in [settings.ORIGINALS_DIR, settings.ARCHIVE_DIR]: 

233 if document_dir.exists() and document_dir.is_dir(): 

234 for entry in document_dir.glob("**/*"): 

235 if entry.is_dir(): 

236 continue 

237 self.stdout.write( 

238 self.style.WARNING( 

239 f"Found file {entry.relative_to(document_dir)}, this might indicate a non-empty installation", 

240 ), 

241 ) 

242 break 

243 # But existing users or other data still matters in a data only 

244 if ( 

245 User.objects.exclude(username__in=["consumer", "AnonymousUser"]).count() 

246 != 0 

247 ): 

248 self.stdout.write( 

249 self.style.WARNING( 

250 "Found existing user(s), this might indicate a non-empty installation", 

251 ), 

252 ) 

253 if Document.global_objects.count() != 0: 

254 self.stdout.write( 

255 self.style.WARNING( 

256 "Found existing documents(s), this might indicate a non-empty installation", 

257 ), 

258 ) 

259 

260 def pre_check_manifest_exists() -> None: 

261 if not (self.source / "manifest.json").exists(): 

262 raise CommandError( 

263 "That directory doesn't appear to contain a manifest.json file.", 

264 ) 

265 

266 if not self.source.exists(): 

267 raise CommandError("That path doesn't exist") 

268 

269 if not os.access(self.source, os.R_OK): 

270 raise CommandError("That path doesn't appear to be readable") 

271 

272 pre_check_maybe_not_empty() 

273 pre_check_manifest_exists() 

274 

275 def load_manifest_files(self) -> None: 

276 """ 

277 Loads manifest data from the various JSON files for parsing and loading the database 

278 """ 

279 main_manifest_path: Path = self.source / "manifest.json" 

280 self.manifest_paths.append(main_manifest_path) 

281 

282 for file in Path(self.source).glob("**/*-manifest.json"): 

283 self.manifest_paths.append(file) 

284 

285 def load_metadata(self) -> None: 

286 """ 

287 Loads either just the version information or the version information and extra data 

288 

289 Must account for the old style of export as well, with just version.json 

290 """ 

291 version_path: Path = self.source / "version.json" 

292 metadata_path: Path = self.source / "metadata.json" 

293 if not version_path.exists() and not metadata_path.exists(): 

294 self.stdout.write( 

295 self.style.NOTICE("No version.json or metadata.json file located"), 

296 ) 

297 return 

298 

299 if metadata_path.exists(): 

300 with metadata_path.open() as infile: 

301 data = json.load(infile) 

302 self.version = data["version"] 

303 if not self.passphrase and EXPORTER_CRYPTO_SETTINGS_NAME in data: 

304 raise CommandError( 

305 "No passphrase was given, but this export contains encrypted fields", 

306 ) 

307 elif EXPORTER_CRYPTO_SETTINGS_NAME in data: 

308 self.load_crypt_params(data) 

309 elif version_path.exists(): 

310 with version_path.open() as infile: 

311 self.version = json.load(infile)["version"] 

312 

313 if self.version and self.version != version.__full_version_str__: 

314 self.stdout.write( 

315 self.style.WARNING( 

316 "Version mismatch: " 

317 f"Currently {version.__full_version_str__}," 

318 f" importing {self.version}." 

319 " Continuing, but import may fail.", 

320 ), 

321 ) 

322 

323 def _finalize_db_load(self, loaded_models: set[type[Model]]) -> None: 

324 """Verify referential integrity and reset auto-increment sequences.""" 

325 through_tables = { 

326 field.remote_field.through._meta.db_table 

327 for model in loaded_models 

328 for field in model._meta.many_to_many 

329 if field.remote_field.through is not None 

330 and field.remote_field.through._meta.auto_created 

331 } 

332 table_names = [m._meta.db_table for m in loaded_models] + list(through_tables) 

333 if table_names: 

334 connection.check_constraints(table_names=table_names) 

335 

336 if loaded_models: 

337 sequence_sql = connection.ops.sequence_reset_sql( 

338 no_style(), 

339 list(loaded_models), 

340 ) 

341 with connection.cursor() as cursor: 

342 for sql in sequence_sql: 

343 cursor.execute(sql) # pragma: no cover 

344 

345 def _import_error_context_message(self) -> str: 

346 """Return a diagnostic string explaining a DB import failure.""" 

347 if ( # pragma: no cover 

348 self.version is not None and self.version != version.__full_version_str__ 

349 ): 

350 return ( # pragma: no cover 

351 "Version mismatch: " 

352 f"Currently {version.__full_version_str__}," 

353 f" importing {self.version}" 

354 ) 

355 return "No version information present" 

356 

357 def load_data_to_database(self) -> None: 

358 """ 

359 Streams records from each manifest path and loads them into the database 

360 using bulk_create with bounded batch sizes, avoiding holding the entire 

361 manifest in memory at once. 

362 

363 Memory bound: at most batch_size * (number of distinct model types 

364 present simultaneously in the manifest) instances at any time. 

365 For the standard non-split manifest, records are grouped by model, so 

366 in practice only one model's batch accumulates at a time. 

367 """ 

368 # Maps model class -> list of (instance, m2m_data) waiting to be flushed 

369 pending: defaultdict[type[Model], list[tuple[Model, M2MData]]] = defaultdict( 

370 list, 

371 ) 

372 # All model classes inserted (needed for sequence reset after the load) 

373 loaded_models: set[type[Model]] = set() 

374 

375 def flush_model(model: type[Model]) -> None: 

376 """bulk_create the pending batch for model, then apply M2M.""" 

377 batch = pending.pop(model, []) 

378 if not batch: # pragma: no cover 

379 return 

380 instances = [inst for inst, _ in batch] 

381 # GeneratedField is excluded because it is generated and trying to insert it will fail 

382 update_fields = [ 

383 f.attname 

384 for f in model._meta.concrete_fields 

385 if not f.primary_key and not isinstance(f, GeneratedField) 

386 ] 

387 if not update_fields: # pragma: no cover 

388 raise DeserializationError( 

389 f"{model.__name__} has no updatable fields; PK-only models are not supported by the importer", 

390 ) 

391 # MySQL/MariaDB support upserts via ON DUPLICATE KEY UPDATE but, 

392 # unlike PostgreSQL/SQLite, cannot target a specific unique field 

393 # for the conflict -- passing unique_fields there raises 

394 # NotSupportedError. 

395 unique_fields = ( 

396 [model._meta.pk.attname] 

397 if connection.features.supports_update_conflicts_with_target 

398 else None 

399 ) 

400 model.objects.bulk_create( # type: ignore[attr-defined] 

401 instances, 

402 update_conflicts=True, 

403 unique_fields=unique_fields, 

404 update_fields=update_fields, 

405 ) 

406 loaded_models.add(model) 

407 for instance, m2m_data in batch: 

408 for field_name, pk_list in m2m_data.items(): 

409 getattr(instance, field_name).set(pk_list) 

410 

411 def flush_all() -> None: 

412 for model in list(pending): 

413 flush_model(model) 

414 

415 try: 

416 with transaction.atomic(): 

417 # ContentType and Permission have auto-assigned PKs on a fresh 

418 # install that conflict with exported PKs. Delete and re-import. 

419 ContentType.objects.all().delete() 

420 Permission.objects.all().delete() 

421 

422 # Constraint checks are disabled so FK/M2M inserts succeed 

423 # regardless of record order within the manifest. 

424 # Note: on SQLite inside a transaction this context manager is a 

425 # no-op; the constraint-deferral path is only exercised on 

426 # PostgreSQL in production. 

427 with connection.constraint_checks_disabled(): 

428 for manifest_path in self.manifest_paths: 

429 for record in iter_manifest_records(manifest_path): 

430 model, instance, m2m_data = _deserialize_record(record) 

431 pending[model].append((instance, m2m_data)) 

432 if len(pending[model]) >= self.batch_size: 

433 flush_model(model) 

434 

435 flush_all() 

436 

437 self._finalize_db_load(loaded_models) 

438 

439 except (FieldDoesNotExist, DeserializationError, IntegrityError): 

440 self.stdout.write(self.style.ERROR("Database import failed")) 

441 self.stdout.write(self.style.ERROR(self._import_error_context_message())) 

442 raise 

443 

444 # ContentType/Permission rows were deleted and reinserted above; stale 

445 # in-process caches must be invalidated so permission checks use the 

446 # new IDs rather than pre-import PKs. 

447 ContentType.objects.clear_cache() 

448 clear_ct_cache() 

449 

450 def handle(self, *args, **options) -> None: 

451 logging.getLogger().handlers[0].level = logging.ERROR 

452 

453 self.source = Path(options["source"]).resolve() 

454 self.data_only: bool = options["data_only"] 

455 self.passphrase: str | None = options.get("passphrase") 

456 self.batch_size: int = options["batch_size"] 

457 self.version: str | None = None 

458 self.salt: str | None = None 

459 self.manifest_paths = [] 

460 

461 # Create a temporary directory for extracting a zip file into it, even if supplied source is no zip file to keep code cleaner. 

462 with tempfile.TemporaryDirectory() as tmp_dir: 

463 if is_zipfile(self.source): 

464 with ZipFile(self.source) as zf: 

465 unsupported = { 

466 info.compress_type 

467 for info in zf.infolist() 

468 if not compress_type_readable(info.compress_type) 

469 } 

470 if unsupported: 

471 names = sorted(unreadable_method_names(unsupported)) 

472 message = ( 

473 f"This archive uses compression this Python version cannot " 

474 f"read ({', '.join(names)})." 

475 ) 

476 if "zstd" in names: 

477 message += " zstd archives require Python 3.14+." 

478 raise CommandError(message) 

479 zf.extractall(tmp_dir) 

480 self.source = Path(tmp_dir) 

481 self._run_import() 

482 

483 def _run_import(self) -> None: 

484 self.pre_check() 

485 self.load_metadata() 

486 self.load_manifest_files() 

487 self.check_manifest_validity() 

488 self.decrypt_secret_fields() 

489 

490 # see /src/documents/signals/handlers.py 

491 with ( 

492 disable_signal( 

493 post_save, 

494 receiver=update_filename_and_move_files, 

495 sender=Document, 

496 weak=False, 

497 ), 

498 disable_signal( 

499 m2m_changed, 

500 receiver=update_filename_and_move_files, 

501 sender=Document.tags.through, 

502 weak=False, 

503 ), 

504 disable_signal( 

505 post_save, 

506 receiver=update_filename_and_move_files, 

507 sender=CustomFieldInstance, 

508 weak=False, 

509 ), 

510 disable_signal( 

511 post_save, 

512 receiver=check_paths_and_prune_custom_fields, 

513 sender=CustomField, 

514 ), 

515 ): 

516 if settings.AUDIT_LOG_ENABLED: 

517 auditlog.unregister(Document) 

518 auditlog.unregister(Correspondent) 

519 auditlog.unregister(Tag) 

520 auditlog.unregister(DocumentType) 

521 auditlog.unregister(Note) 

522 auditlog.unregister(CustomField) 

523 auditlog.unregister(CustomFieldInstance) 

524 

525 # Fill up the database with whatever is in the manifest 

526 self.load_data_to_database() 

527 

528 if not self.data_only: 

529 self._import_files_from_manifest() 

530 else: 

531 self.stdout.write(self.style.NOTICE("Data only import completed")) 

532 

533 for tmp in getattr(self, "_decrypted_tmp_paths", []): 

534 tmp.unlink(missing_ok=True) 

535 

536 self.stdout.write("Updating search index...") 

537 call_command( 

538 "document_index", 

539 "reindex", 

540 no_progress_bar=self.no_progress_bar, 

541 ) 

542 

543 def check_manifest_validity(self) -> None: 

544 """ 

545 Attempts to verify the manifest is valid. Namely checking the files 

546 referred to exist and the files can be read from 

547 """ 

548 

549 def check_document_validity(document_record: dict) -> None: 

550 if EXPORTER_FILE_NAME not in document_record: 

551 raise CommandError( 

552 "The manifest file contains a record which does not " 

553 "refer to an actual document file.", 

554 ) 

555 

556 doc_file = document_record[EXPORTER_FILE_NAME] 

557 doc_path: Path = self.source / doc_file 

558 if not doc_path.exists(): 

559 raise CommandError( 

560 f'The manifest file refers to "{doc_file}" which does not ' 

561 "appear to be in the source directory.", 

562 ) 

563 try: 

564 with doc_path.open(mode="rb"): 

565 pass 

566 except Exception as e: 

567 raise CommandError( 

568 f"Failed to read from original file {doc_path}", 

569 ) from e 

570 

571 if EXPORTER_ARCHIVE_NAME in document_record: 

572 archive_file = document_record[EXPORTER_ARCHIVE_NAME] 

573 doc_archive_path: Path = self.source / archive_file 

574 if not doc_archive_path.exists(): 

575 raise CommandError( 

576 f"The manifest file refers to {archive_file} which " 

577 f"does not appear to be in the source directory.", 

578 ) 

579 try: 

580 with doc_archive_path.open(mode="rb"): 

581 pass 

582 except Exception as e: 

583 raise CommandError( 

584 f"Failed to read from archive file {doc_archive_path}", 

585 ) from e 

586 

587 def check_share_link_bundle_validity(bundle_record: dict) -> None: 

588 if EXPORTER_SHARE_LINK_BUNDLE_NAME not in bundle_record: 

589 return 

590 

591 bundle_file = bundle_record[EXPORTER_SHARE_LINK_BUNDLE_NAME] 

592 bundle_path: Path = self.source / bundle_file 

593 if not bundle_path.exists(): 

594 raise CommandError( 

595 f'The manifest file refers to "{bundle_file}" which does not ' 

596 "appear to be in the source directory.", 

597 ) 

598 try: 

599 with bundle_path.open(mode="rb"): 

600 pass 

601 except Exception as e: 

602 raise CommandError( 

603 f"Failed to read from share link bundle file {bundle_path}", 

604 ) from e 

605 

606 self.stdout.write("Checking the manifest") 

607 for manifest_path in self.manifest_paths: 

608 for record in iter_manifest_records(manifest_path): 

609 # Only check if the document files exist if this is not data only 

610 # We don't care about documents for a data only import 

611 if self.data_only: 

612 continue 

613 if record["model"] == "documents.document": 

614 check_document_validity(record) 

615 elif record["model"] == "documents.sharelinkbundle": 

616 check_share_link_bundle_validity(record) 

617 

618 def _import_files_from_manifest(self) -> None: 

619 settings.ORIGINALS_DIR.mkdir(parents=True, exist_ok=True) 

620 settings.THUMBNAIL_DIR.mkdir(parents=True, exist_ok=True) 

621 settings.ARCHIVE_DIR.mkdir(parents=True, exist_ok=True) 

622 settings.SHARE_LINK_BUNDLE_DIR.mkdir(parents=True, exist_ok=True) 

623 

624 self.stdout.write("Copy files into paperless...") 

625 

626 for record in self.track( 

627 _iter_document_copy_records(self.manifest_paths), 

628 description="Copying files...", 

629 ): 

630 document = Document.global_objects.get(pk=record["pk"]) 

631 

632 doc_file = record[EXPORTER_FILE_NAME] 

633 document_path = self.source / doc_file 

634 

635 if record[EXPORTER_THUMBNAIL_NAME]: 

636 thumb_file = record[EXPORTER_THUMBNAIL_NAME] 

637 thumbnail_path = (self.source / thumb_file).resolve() 

638 else: 

639 thumbnail_path = None 

640 

641 if record[EXPORTER_ARCHIVE_NAME]: 

642 archive_file = record[EXPORTER_ARCHIVE_NAME] 

643 archive_path = self.source / archive_file 

644 else: 

645 archive_path = None 

646 

647 with FileLock(settings.MEDIA_LOCK): 

648 if Path(document.source_path).is_file(): 

649 raise FileExistsError(document.source_path) 

650 

651 create_source_path_directory(document.source_path) 

652 

653 copy_file_with_basic_stats(document_path, document.source_path) 

654 

655 if thumbnail_path: 

656 copy_file_with_basic_stats( 

657 thumbnail_path, 

658 document.thumbnail_path, 

659 ) 

660 

661 if archive_path: 

662 create_source_path_directory(document.archive_path) 

663 # TODO: this assumes that the export is valid and 

664 # archive_filename is present on all documents with 

665 # archived files 

666 copy_file_with_basic_stats(archive_path, document.archive_path) 

667 

668 for record in self.track( 

669 _iter_share_link_bundle_copy_records(self.manifest_paths), 

670 description="Copying share link bundles...", 

671 ): 

672 bundle = ShareLinkBundle.objects.get(pk=record["pk"]) 

673 bundle_file = record[EXPORTER_SHARE_LINK_BUNDLE_NAME] 

674 bundle_source_path = (self.source / bundle_file).resolve() 

675 bundle_target_path = bundle.absolute_file_path 

676 if bundle_target_path is None: 

677 raise CommandError( 

678 f"Share link bundle {bundle.pk} does not have a valid file path.", 

679 ) 

680 

681 with FileLock(settings.MEDIA_LOCK): 

682 bundle_target_path.parent.mkdir(parents=True, exist_ok=True) 

683 copy_file_with_basic_stats( 

684 bundle_source_path, 

685 bundle_target_path, 

686 ) 

687 

688 def _decrypt_record_if_needed(self, record: dict) -> dict: 

689 fields = self.CRYPT_FIELDS_BY_MODEL.get(record.get("model", "")) 

690 if fields: 

691 for field in fields: 

692 if record["fields"].get(field): 

693 record["fields"][field] = self.decrypt_string( 

694 value=record["fields"][field], 

695 ) 

696 return record 

697 

698 def decrypt_secret_fields(self) -> None: 

699 """ 

700 The converse decryption of some fields out of the export before importing to database. 

701 Streams records from each manifest path and writes decrypted content to a temp file. 

702 """ 

703 if not self.passphrase: 

704 return 

705 # Salt has been loaded from metadata.json at this point, so it cannot be None 

706 self.setup_crypto(passphrase=self.passphrase, salt=self.salt) 

707 self._decrypted_tmp_paths: list[Path] = [] 

708 new_paths: list[Path] = [] 

709 for manifest_path in self.manifest_paths: 

710 tmp = manifest_path.with_name(manifest_path.stem + ".decrypted.json") 

711 with tmp.open("w", encoding="utf-8") as out: 

712 out.write("[\n") 

713 first = True 

714 for record in iter_manifest_records(manifest_path): 

715 if not first: 

716 out.write(",\n") 

717 json.dump( 

718 self._decrypt_record_if_needed(record), 

719 out, 

720 indent=2, 

721 ensure_ascii=False, 

722 ) 

723 first = False 

724 out.write("\n]\n") 

725 self._decrypted_tmp_paths.append(tmp) 

726 new_paths.append(tmp) 

727 self.manifest_paths = new_paths