Coverage for documents/barcodes.py: 23%

231 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import logging 

4import re 

5import tempfile 

6from dataclasses import dataclass 

7from pathlib import Path 

8from typing import TYPE_CHECKING 

9 

10import regex as regex_mod 

11from django.conf import settings 

12from pdf2image import convert_from_path 

13from pikepdf import Page 

14from pikepdf import PasswordError 

15from pikepdf import Pdf 

16 

17from documents.converters import convert_from_tiff_to_pdf 

18from documents.data_models import ConsumableDocument 

19from documents.data_models import DocumentMetadataOverrides 

20from documents.data_models import DocumentSource 

21from documents.data_models import StoredBarcode 

22from documents.models import Document 

23from documents.models import PaperlessTask 

24from documents.models import Tag 

25from documents.plugins.base import ConsumeTaskPlugin 

26from documents.plugins.base import StopConsumeTaskError 

27from documents.plugins.helpers import ProgressManager 

28from documents.plugins.helpers import ProgressStatusOptions 

29from documents.regex import safe_regex_match 

30from documents.regex import safe_regex_sub 

31from documents.utils import copy_basic_file_stats 

32from documents.utils import copy_file_with_basic_stats 

33from documents.utils import maybe_override_pixel_limit 

34from paperless.config import BarcodeConfig 

35 

36if TYPE_CHECKING: 36 ↛ 37line 36 didn't jump to line 37 because the condition on line 36 was never true

37 from PIL import Image 

38 

39logger = logging.getLogger("paperless.barcodes") 

40 

41 

42@dataclass(frozen=True) 

43class Barcode: 

44 """ 

45 Holds the information about a single barcode and its location in a document 

46 """ 

47 

48 page: int 

49 value: str 

50 settings: BarcodeConfig 

51 format: str = "" 

52 

53 @property 

54 def is_separator(self) -> bool: 

55 """ 

56 Returns True if the barcode value equals the configured separation value, 

57 False otherwise 

58 """ 

59 return self.value == self.settings.barcode_string 

60 

61 @property 

62 def is_asn(self) -> bool: 

63 """ 

64 Returns True if the barcode value matches the configured ASN prefix, 

65 False otherwise 

66 """ 

67 return self.value.startswith(self.settings.barcode_asn_prefix) 

68 

69 @property 

70 def is_tag(self) -> bool: 

71 """ 

72 Returns True if the barcode value matches any configured tag mapping pattern, 

73 False otherwise. 

74 

75 Note: This does NOT exclude ASN or separator barcodes - they can also be used 

76 as tags if they match a tag mapping pattern (e.g., {"ASN12.*": "JOHN"}). 

77 """ 

78 for pattern in self.settings.barcode_tag_mapping: 

79 if safe_regex_match(pattern, self.value, flags=regex_mod.IGNORECASE): 

80 return True 

81 return False 

82 

83 def stored(self) -> StoredBarcode: 

84 """ 

85 The barcode as it is stored with a document, page 1-indexed 

86 """ 

87 return {"page": self.page + 1, "value": self.value, "format": self.format} 

88 

89 

90class BarcodePlugin(ConsumeTaskPlugin): 

91 NAME: str = "BarcodePlugin" 

92 

93 @property 

94 def able_to_run(self) -> bool: 

95 """ 

96 Able to run if: 

97 - ASN from barcode detection is enabled or 

98 - Barcode support is enabled and the mime type is supported 

99 """ 

100 return ( 

101 self.settings.barcode_enable_asn 

102 or self.settings.barcodes_enabled 

103 or self.settings.barcode_enable_tag 

104 or self.settings.barcode_store_values 

105 ) and self.input_doc.mime_type in scannable_mime_types(self.settings) 

106 

107 def get_settings(self) -> BarcodeConfig: 

108 """ 

109 Returns the settings for this plugin (Django settings or app config) 

110 """ 

111 return BarcodeConfig() 

112 

113 def __init__( 

114 self, 

115 input_doc: ConsumableDocument, 

116 metadata: DocumentMetadataOverrides, 

117 status_mgr: ProgressManager, 

118 base_tmp_dir: Path, 

119 task_id: str, 

120 ) -> None: 

121 super().__init__( 

122 input_doc, 

123 metadata, 

124 status_mgr, 

125 base_tmp_dir, 

126 task_id, 

127 ) 

128 # need these for able_to_run 

129 self.settings = self.get_settings() 

130 

131 def setup(self) -> None: 

132 self.temp_dir = tempfile.TemporaryDirectory( 

133 dir=self.base_tmp_dir, 

134 prefix="barcode", 

135 ) 

136 self.pdf_file: Path = self.input_doc.original_file 

137 self._tiff_conversion_done = False 

138 self.barcodes: list[Barcode] = [] 

139 

140 def _apply_detected_asn(self, detected_asn: int) -> None: 

141 """ 

142 Apply a detected ASN to metadata if allowed. 

143 """ 

144 if ( 

145 self.metadata.skip_asn_if_exists 

146 and Document.global_objects.filter( 

147 archive_serial_number=detected_asn, 

148 ).exists() 

149 ): 

150 logger.info( 

151 f"Found ASN in barcode {detected_asn} but skipping because it already exists.", 

152 ) 

153 return 

154 

155 logger.info(f"Found ASN in barcode: {detected_asn}") 

156 self.metadata.asn = detected_asn 

157 

158 def run(self) -> None: 

159 # Some operations may use PIL, override pixel setting if needed 

160 maybe_override_pixel_limit() 

161 

162 # Maybe do the conversion of TIFF to PDF 

163 self.convert_from_tiff_to_pdf() 

164 

165 # Locate any barcodes in the files 

166 self.detect() 

167 

168 # try reading tags from barcodes 

169 # If tag splitting is enabled, skip this on the original document - let each split document extract its own tags 

170 # However, if we're processing a split document (original_path is set), extract tags 

171 if ( 

172 self.settings.barcode_enable_tag 

173 and ( 

174 not self.settings.barcode_tag_split 

175 or self.input_doc.original_path is not None 

176 ) 

177 and (tags := self.tags) is not None 

178 and len(tags) > 0 

179 ): 

180 if self.metadata.tag_ids: 

181 self.metadata.tag_ids += tags 

182 else: 

183 self.metadata.tag_ids = tags 

184 logger.info(f"Found tags in barcode: {tags}") 

185 

186 # Lastly attempt to split documents 

187 if self.settings.barcodes_enabled and ( 

188 separator_pages := self.get_separation_pages() 

189 ): 

190 # We have pages to split against 

191 

192 # Note this does NOT use the base_temp_dir, as that will be removed 

193 tmp_dir = Path( 

194 tempfile.mkdtemp( 

195 dir=settings.SCRATCH_DIR, 

196 prefix="paperless-barcode-split-", 

197 ), 

198 ).resolve() 

199 

200 from documents import tasks 

201 

202 _SOURCE_TO_TRIGGER: dict[DocumentSource, PaperlessTask.TriggerSource] = { 

203 DocumentSource.ConsumeFolder: PaperlessTask.TriggerSource.FOLDER_CONSUME, 

204 DocumentSource.ApiUpload: PaperlessTask.TriggerSource.API_UPLOAD, 

205 DocumentSource.MailFetch: PaperlessTask.TriggerSource.EMAIL_CONSUME, 

206 DocumentSource.WebUI: PaperlessTask.TriggerSource.WEB_UI, 

207 } 

208 trigger_source = _SOURCE_TO_TRIGGER.get( 

209 self.input_doc.source, 

210 PaperlessTask.TriggerSource.MANUAL, 

211 ) 

212 

213 # Create the split document tasks 

214 for new_document in self.separate_pages(separator_pages): 

215 copy_file_with_basic_stats(new_document, tmp_dir / new_document.name) 

216 

217 task = tasks.consume_file.apply_async( 

218 kwargs={ 

219 "input_doc": ConsumableDocument( 

220 # Same source, for templates 

221 source=self.input_doc.source, 

222 mailrule_id=self.input_doc.mailrule_id, 

223 # Can't use same folder or the consume might grab it again 

224 original_file=(tmp_dir / new_document.name).resolve(), 

225 # Adding optional original_path for later uses in 

226 # workflow matching 

227 original_path=self.input_doc.original_file, 

228 ), 

229 "overrides": self.metadata, 

230 }, 

231 headers={"trigger_source": trigger_source}, 

232 ) 

233 logger.info(f"Created new task {task.id} for {new_document.name}") 

234 

235 # This file is now two or more files 

236 self.input_doc.original_file.unlink() 

237 

238 msg = "Barcode splitting complete!" 

239 

240 # Update the progress to complete 

241 self.status_mgr.send_progress(ProgressStatusOptions.SUCCESS, msg, 100, 100) 

242 

243 # Request the consume task stops 

244 raise StopConsumeTaskError(msg) 

245 

246 # Update/overwrite an ASN if possible 

247 # After splitting, as otherwise each split document gets the same ASN 

248 if self.settings.barcode_enable_asn and (located_asn := self.asn) is not None: 

249 self._apply_detected_asn(located_asn) 

250 

251 # After splitting too, so each split document keeps its own barcodes 

252 if self.settings.barcode_store_values: 

253 self.metadata.barcodes = [x.stored() for x in self.barcodes] or None 

254 

255 def cleanup(self) -> None: 

256 self.temp_dir.cleanup() 

257 

258 def convert_from_tiff_to_pdf(self) -> None: 

259 """ 

260 May convert a TIFF image into a PDF, if the input is a TIFF and 

261 the TIFF has not been made into a PDF 

262 """ 

263 # Nothing to do, pdf_file is already assigned correctly 

264 if self.input_doc.mime_type != "image/tiff" or self._tiff_conversion_done: 

265 return 

266 

267 self.pdf_file = convert_from_tiff_to_pdf( 

268 self.input_doc.original_file, 

269 Path(self.temp_dir.name), 

270 ) 

271 self._tiff_conversion_done = True 

272 

273 def detect(self) -> None: 

274 """ 

275 Scan all pages of the PDF as images, updating barcodes and the pages 

276 found on as we go 

277 """ 

278 # Bail if barcodes already exist 

279 if self.barcodes: 

280 return 

281 

282 # No op if not a TIFF 

283 self.convert_from_tiff_to_pdf() 

284 

285 try: 

286 self.barcodes = scan_pdf( 

287 self.pdf_file, 

288 self.settings, 

289 Path(self.temp_dir.name), 

290 ) 

291 

292 # Password protected files can't be checked 

293 # This is the exception raised for those 

294 except PasswordError as e: 

295 logger.warning( 

296 f"File is likely password protected, not checking for barcodes: {e}", 

297 ) 

298 # This file is really borked, allow the consumption to continue 

299 # but it may fail further on 

300 except Exception as e: # pragma: no cover 

301 logger.warning( 

302 f"Exception during barcode scanning: {e}", 

303 ) 

304 

305 @property 

306 def asn(self) -> int | None: 

307 """ 

308 Search the parsed barcodes for any ASNs. 

309 The first barcode that starts with barcode_asn_prefix 

310 is considered the ASN to be used. 

311 Returns the detected ASN (or None) 

312 """ 

313 asn = None 

314 

315 # Ensure the barcodes have been read 

316 self.detect() 

317 

318 # get the first barcode that starts with barcode_asn_prefix 

319 asn_text: str | None = next( 

320 (x.value for x in self.barcodes if x.is_asn), 

321 None, 

322 ) 

323 

324 if asn_text: 

325 logger.debug(f"Found ASN Barcode: {asn_text}") 

326 # remove the prefix and remove whitespace 

327 asn_text = asn_text[len(self.settings.barcode_asn_prefix) :].strip() 

328 

329 # remove non-numeric parts of the remaining string 

330 asn_text = re.sub(r"\D", "", asn_text) 

331 

332 # now, try parsing the ASN number 

333 try: 

334 asn = int(asn_text) 

335 except ValueError as e: 

336 logger.warning(f"Failed to parse ASN number because: {e}") 

337 

338 return asn 

339 

340 @property 

341 def tags(self) -> list[int]: 

342 """ 

343 Search the parsed barcodes for any tags. 

344 Returns the detected tag ids (or empty list) 

345 """ 

346 tags: list[int] = [] 

347 

348 # Ensure the barcodes have been read 

349 self.detect() 

350 

351 for x in self.barcodes: 

352 tag_texts: str = x.value 

353 

354 for raw in tag_texts.split(","): 

355 try: 

356 tag_str: str | None = None 

357 for pattern in self.settings.barcode_tag_mapping: 

358 if safe_regex_match(pattern, raw, flags=regex_mod.IGNORECASE): 

359 sub = self.settings.barcode_tag_mapping[pattern] 

360 tag_str = ( 

361 safe_regex_sub( 

362 pattern, 

363 sub, 

364 raw, 

365 flags=regex_mod.IGNORECASE, 

366 ) 

367 if sub 

368 else raw 

369 ) 

370 break 

371 

372 if tag_str: 

373 tag, _ = Tag.objects.get_or_create( 

374 name__iexact=tag_str, 

375 defaults={"name": tag_str}, 

376 ) 

377 

378 logger.debug( 

379 f"Found Tag Barcode '{raw}', substituted " 

380 f"to '{tag}' and mapped to " 

381 f"tag #{tag.pk}.", 

382 ) 

383 tags.append(tag.pk) 

384 

385 except Exception as e: 

386 logger.error( 

387 f"Failed to find or create TAG '{raw}' because: {e}", 

388 ) 

389 

390 return tags 

391 

392 def get_separation_pages(self) -> dict[int, bool]: 

393 """ 

394 Search the parsed barcodes for separators and returns a dict of page 

395 numbers, which separate the file into new files, together with the 

396 information whether to keep the page. 

397 """ 

398 # filter all barcodes for the separator string 

399 # get the page numbers of the separating barcodes 

400 retain = self.settings.barcode_retain_split_pages 

401 separator_pages = { 

402 bc.page: retain 

403 for bc in self.barcodes 

404 if bc.is_separator and (not retain or (retain and bc.page > 0)) 

405 } # as below, dont include the first page if retain is enabled 

406 

407 # add the page numbers of the ASN barcodes 

408 # (except for first page, that might lead to infinite loops). 

409 if self.settings.barcode_enable_asn: 

410 separator_pages = { 

411 **separator_pages, 

412 **{bc.page: True for bc in self.barcodes if bc.is_asn and bc.page != 0}, 

413 } 

414 

415 # add the page numbers of the TAG barcodes if splitting is enabled 

416 # (except for first page, that might lead to infinite loops). 

417 if self.settings.barcode_tag_split and self.settings.barcode_enable_tag: 

418 separator_pages = { 

419 **separator_pages, 

420 **{bc.page: True for bc in self.barcodes if bc.is_tag and bc.page != 0}, 

421 } 

422 

423 return separator_pages 

424 

425 def separate_pages(self, pages_to_split_on: dict[int, bool]) -> list[Path]: 

426 """ 

427 Separate the provided pdf file on the pages_to_split_on. 

428 The pages which are defined by the keys in page_numbers 

429 will be removed if the corresponding value is false. 

430 Returns a list of (temporary) filepaths to consume. 

431 These will need to be deleted later. 

432 """ 

433 

434 document_paths = [] 

435 fname: str = self.input_doc.original_file.stem 

436 with Pdf.open(self.pdf_file) as input_pdf: 

437 # Start with an empty document 

438 current_document: list[Page] = [] 

439 # A list of documents, ie a list of lists of pages 

440 documents: list[list[Page]] = [current_document] 

441 

442 for idx, page in enumerate(input_pdf.pages): 

443 # Keep building the new PDF as long as it is not a 

444 # separator index 

445 if idx not in pages_to_split_on: 

446 current_document.append(page) 

447 continue 

448 

449 # This is a split index 

450 # Start a new destination page listing 

451 logger.debug(f"Starting new document at idx {idx}") 

452 current_document = [] 

453 documents.append(current_document) 

454 keep_page: bool = pages_to_split_on[idx] 

455 if keep_page: 

456 # Keep the page 

457 # (new document is started by asn barcode) 

458 current_document.append(page) 

459 

460 documents = [x for x in documents if len(x)] 

461 

462 logger.debug(f"Split into {len(documents)} new documents") 

463 

464 # Write the new documents out 

465 for doc_idx, document in enumerate(documents): 

466 dst = Pdf.new() 

467 dst.pages.extend(document) 

468 

469 output_filename = f"{fname}_document_{doc_idx}.pdf" 

470 

471 logger.debug(f"pdf no:{doc_idx} has {len(dst.pages)} pages") 

472 savepath = Path(self.temp_dir.name) / output_filename 

473 with savepath.open("wb") as out: 

474 dst.save(out) 

475 

476 copy_basic_file_stats(self.input_doc.original_file, savepath) 

477 

478 document_paths.append(savepath) 

479 

480 return document_paths 

481 

482 

483def scannable_mime_types(settings: BarcodeConfig) -> set[str]: 

484 """ 

485 The file types the barcode scan supports with the current settings 

486 """ 

487 if settings.barcode_enable_tiff_support: 

488 return {"application/pdf", "image/tiff"} 

489 return {"application/pdf"} 

490 

491 

492def read_barcodes_zxing(image: Image.Image) -> list[tuple[str, str]]: 

493 """ 

494 Returns the text and format (zxing enum name) of each barcode found in 

495 the image 

496 """ 

497 barcodes = [] 

498 

499 import zxingcpp 

500 

501 detected_barcodes = zxingcpp.read_barcodes(image) 

502 for barcode in detected_barcodes: 

503 if barcode.text: 

504 barcodes.append((barcode.text, barcode.format.name)) 

505 logger.debug( 

506 f"Barcode of type {barcode.format} found: {barcode.text}", 

507 ) 

508 

509 return barcodes 

510 

511 

512def scan_pdf(pdf_path: Path, settings: BarcodeConfig, work_dir: Path) -> list[Barcode]: 

513 """ 

514 Scans the pages of a PDF as images for barcodes. Errors are not caught, 

515 so callers can tell a failed scan from one that found nothing. 

516 """ 

517 barcodes: list[Barcode] = [] 

518 

519 with Pdf.open(pdf_path) as pdf: 

520 num_of_pages = len(pdf.pages) 

521 logger.debug(f"PDF has {num_of_pages} pages") 

522 

523 # Get limit from configuration 

524 barcode_max_pages: int = ( 

525 num_of_pages if settings.barcode_max_pages == 0 else settings.barcode_max_pages 

526 ) 

527 

528 if barcode_max_pages < num_of_pages: # pragma: no cover 

529 logger.debug( 

530 f"Barcodes detection will be limited to the first {barcode_max_pages} pages", 

531 ) 

532 

533 for current_page_number in range(min(num_of_pages, barcode_max_pages)): 

534 logger.debug(f"Processing page {current_page_number}") 

535 

536 # Convert page to image 

537 page = convert_from_path( 

538 pdf_path, 

539 dpi=settings.barcode_dpi, 

540 output_folder=work_dir, 

541 first_page=current_page_number + 1, 

542 last_page=current_page_number + 1, 

543 )[0] 

544 

545 # Remember filename, since it is lost by upscaling 

546 page_filepath = Path(page.filename) 

547 logger.debug(f"Image is at {page_filepath}") 

548 

549 # Upscale image if configured 

550 factor = settings.barcode_upscale 

551 if factor > 1.0: 

552 logger.debug( 

553 f"Upscaling image by {factor} for better barcode detection", 

554 ) 

555 x, y = page.size 

556 page = page.resize( 

557 (round(x * factor), (round(y * factor))), 

558 ) 

559 

560 for barcode_value, barcode_format in read_barcodes_zxing(page): 

561 barcodes.append( 

562 Barcode(current_page_number, barcode_value, settings, barcode_format), 

563 ) 

564 

565 # Delete temporary image file 

566 page_filepath.unlink() 

567 

568 return barcodes 

569 

570 

571def read_barcode_values( 

572 path: Path, 

573 mime_type: str, 

574 settings: BarcodeConfig, 

575 work_dir: Path, 

576) -> list[StoredBarcode] | None: 

577 """ 

578 Reads the barcodes of a file outside of the consumption plugins: for new 

579 versions, which skip the barcode plugin, and when reprocessing. 

580 

581 Returns None if the file can't be scanned with the current settings. 

582 Errors while scanning are raised. 

583 """ 

584 if mime_type not in scannable_mime_types(settings): 

585 return None 

586 if mime_type == "image/tiff": 

587 path = convert_from_tiff_to_pdf(path, work_dir) 

588 return [x.stored() for x in scan_pdf(path, settings, work_dir)]