Coverage for documents/barcodes.py: 23%
231 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import logging
4import re
5import tempfile
6from dataclasses import dataclass
7from pathlib import Path
8from typing import TYPE_CHECKING
10import regex as regex_mod
11from django.conf import settings
12from pdf2image import convert_from_path
13from pikepdf import Page
14from pikepdf import PasswordError
15from pikepdf import Pdf
17from documents.converters import convert_from_tiff_to_pdf
18from documents.data_models import ConsumableDocument
19from documents.data_models import DocumentMetadataOverrides
20from documents.data_models import DocumentSource
21from documents.data_models import StoredBarcode
22from documents.models import Document
23from documents.models import PaperlessTask
24from documents.models import Tag
25from documents.plugins.base import ConsumeTaskPlugin
26from documents.plugins.base import StopConsumeTaskError
27from documents.plugins.helpers import ProgressManager
28from documents.plugins.helpers import ProgressStatusOptions
29from documents.regex import safe_regex_match
30from documents.regex import safe_regex_sub
31from documents.utils import copy_basic_file_stats
32from documents.utils import copy_file_with_basic_stats
33from documents.utils import maybe_override_pixel_limit
34from paperless.config import BarcodeConfig
36if TYPE_CHECKING: 36 ↛ 37line 36 didn't jump to line 37 because the condition on line 36 was never true
37 from PIL import Image
39logger = logging.getLogger("paperless.barcodes")
42@dataclass(frozen=True)
43class Barcode:
44 """
45 Holds the information about a single barcode and its location in a document
46 """
48 page: int
49 value: str
50 settings: BarcodeConfig
51 format: str = ""
53 @property
54 def is_separator(self) -> bool:
55 """
56 Returns True if the barcode value equals the configured separation value,
57 False otherwise
58 """
59 return self.value == self.settings.barcode_string
61 @property
62 def is_asn(self) -> bool:
63 """
64 Returns True if the barcode value matches the configured ASN prefix,
65 False otherwise
66 """
67 return self.value.startswith(self.settings.barcode_asn_prefix)
69 @property
70 def is_tag(self) -> bool:
71 """
72 Returns True if the barcode value matches any configured tag mapping pattern,
73 False otherwise.
75 Note: This does NOT exclude ASN or separator barcodes - they can also be used
76 as tags if they match a tag mapping pattern (e.g., {"ASN12.*": "JOHN"}).
77 """
78 for pattern in self.settings.barcode_tag_mapping:
79 if safe_regex_match(pattern, self.value, flags=regex_mod.IGNORECASE):
80 return True
81 return False
83 def stored(self) -> StoredBarcode:
84 """
85 The barcode as it is stored with a document, page 1-indexed
86 """
87 return {"page": self.page + 1, "value": self.value, "format": self.format}
90class BarcodePlugin(ConsumeTaskPlugin):
91 NAME: str = "BarcodePlugin"
93 @property
94 def able_to_run(self) -> bool:
95 """
96 Able to run if:
97 - ASN from barcode detection is enabled or
98 - Barcode support is enabled and the mime type is supported
99 """
100 return (
101 self.settings.barcode_enable_asn
102 or self.settings.barcodes_enabled
103 or self.settings.barcode_enable_tag
104 or self.settings.barcode_store_values
105 ) and self.input_doc.mime_type in scannable_mime_types(self.settings)
107 def get_settings(self) -> BarcodeConfig:
108 """
109 Returns the settings for this plugin (Django settings or app config)
110 """
111 return BarcodeConfig()
113 def __init__(
114 self,
115 input_doc: ConsumableDocument,
116 metadata: DocumentMetadataOverrides,
117 status_mgr: ProgressManager,
118 base_tmp_dir: Path,
119 task_id: str,
120 ) -> None:
121 super().__init__(
122 input_doc,
123 metadata,
124 status_mgr,
125 base_tmp_dir,
126 task_id,
127 )
128 # need these for able_to_run
129 self.settings = self.get_settings()
131 def setup(self) -> None:
132 self.temp_dir = tempfile.TemporaryDirectory(
133 dir=self.base_tmp_dir,
134 prefix="barcode",
135 )
136 self.pdf_file: Path = self.input_doc.original_file
137 self._tiff_conversion_done = False
138 self.barcodes: list[Barcode] = []
140 def _apply_detected_asn(self, detected_asn: int) -> None:
141 """
142 Apply a detected ASN to metadata if allowed.
143 """
144 if (
145 self.metadata.skip_asn_if_exists
146 and Document.global_objects.filter(
147 archive_serial_number=detected_asn,
148 ).exists()
149 ):
150 logger.info(
151 f"Found ASN in barcode {detected_asn} but skipping because it already exists.",
152 )
153 return
155 logger.info(f"Found ASN in barcode: {detected_asn}")
156 self.metadata.asn = detected_asn
158 def run(self) -> None:
159 # Some operations may use PIL, override pixel setting if needed
160 maybe_override_pixel_limit()
162 # Maybe do the conversion of TIFF to PDF
163 self.convert_from_tiff_to_pdf()
165 # Locate any barcodes in the files
166 self.detect()
168 # try reading tags from barcodes
169 # If tag splitting is enabled, skip this on the original document - let each split document extract its own tags
170 # However, if we're processing a split document (original_path is set), extract tags
171 if (
172 self.settings.barcode_enable_tag
173 and (
174 not self.settings.barcode_tag_split
175 or self.input_doc.original_path is not None
176 )
177 and (tags := self.tags) is not None
178 and len(tags) > 0
179 ):
180 if self.metadata.tag_ids:
181 self.metadata.tag_ids += tags
182 else:
183 self.metadata.tag_ids = tags
184 logger.info(f"Found tags in barcode: {tags}")
186 # Lastly attempt to split documents
187 if self.settings.barcodes_enabled and (
188 separator_pages := self.get_separation_pages()
189 ):
190 # We have pages to split against
192 # Note this does NOT use the base_temp_dir, as that will be removed
193 tmp_dir = Path(
194 tempfile.mkdtemp(
195 dir=settings.SCRATCH_DIR,
196 prefix="paperless-barcode-split-",
197 ),
198 ).resolve()
200 from documents import tasks
202 _SOURCE_TO_TRIGGER: dict[DocumentSource, PaperlessTask.TriggerSource] = {
203 DocumentSource.ConsumeFolder: PaperlessTask.TriggerSource.FOLDER_CONSUME,
204 DocumentSource.ApiUpload: PaperlessTask.TriggerSource.API_UPLOAD,
205 DocumentSource.MailFetch: PaperlessTask.TriggerSource.EMAIL_CONSUME,
206 DocumentSource.WebUI: PaperlessTask.TriggerSource.WEB_UI,
207 }
208 trigger_source = _SOURCE_TO_TRIGGER.get(
209 self.input_doc.source,
210 PaperlessTask.TriggerSource.MANUAL,
211 )
213 # Create the split document tasks
214 for new_document in self.separate_pages(separator_pages):
215 copy_file_with_basic_stats(new_document, tmp_dir / new_document.name)
217 task = tasks.consume_file.apply_async(
218 kwargs={
219 "input_doc": ConsumableDocument(
220 # Same source, for templates
221 source=self.input_doc.source,
222 mailrule_id=self.input_doc.mailrule_id,
223 # Can't use same folder or the consume might grab it again
224 original_file=(tmp_dir / new_document.name).resolve(),
225 # Adding optional original_path for later uses in
226 # workflow matching
227 original_path=self.input_doc.original_file,
228 ),
229 "overrides": self.metadata,
230 },
231 headers={"trigger_source": trigger_source},
232 )
233 logger.info(f"Created new task {task.id} for {new_document.name}")
235 # This file is now two or more files
236 self.input_doc.original_file.unlink()
238 msg = "Barcode splitting complete!"
240 # Update the progress to complete
241 self.status_mgr.send_progress(ProgressStatusOptions.SUCCESS, msg, 100, 100)
243 # Request the consume task stops
244 raise StopConsumeTaskError(msg)
246 # Update/overwrite an ASN if possible
247 # After splitting, as otherwise each split document gets the same ASN
248 if self.settings.barcode_enable_asn and (located_asn := self.asn) is not None:
249 self._apply_detected_asn(located_asn)
251 # After splitting too, so each split document keeps its own barcodes
252 if self.settings.barcode_store_values:
253 self.metadata.barcodes = [x.stored() for x in self.barcodes] or None
255 def cleanup(self) -> None:
256 self.temp_dir.cleanup()
258 def convert_from_tiff_to_pdf(self) -> None:
259 """
260 May convert a TIFF image into a PDF, if the input is a TIFF and
261 the TIFF has not been made into a PDF
262 """
263 # Nothing to do, pdf_file is already assigned correctly
264 if self.input_doc.mime_type != "image/tiff" or self._tiff_conversion_done:
265 return
267 self.pdf_file = convert_from_tiff_to_pdf(
268 self.input_doc.original_file,
269 Path(self.temp_dir.name),
270 )
271 self._tiff_conversion_done = True
273 def detect(self) -> None:
274 """
275 Scan all pages of the PDF as images, updating barcodes and the pages
276 found on as we go
277 """
278 # Bail if barcodes already exist
279 if self.barcodes:
280 return
282 # No op if not a TIFF
283 self.convert_from_tiff_to_pdf()
285 try:
286 self.barcodes = scan_pdf(
287 self.pdf_file,
288 self.settings,
289 Path(self.temp_dir.name),
290 )
292 # Password protected files can't be checked
293 # This is the exception raised for those
294 except PasswordError as e:
295 logger.warning(
296 f"File is likely password protected, not checking for barcodes: {e}",
297 )
298 # This file is really borked, allow the consumption to continue
299 # but it may fail further on
300 except Exception as e: # pragma: no cover
301 logger.warning(
302 f"Exception during barcode scanning: {e}",
303 )
305 @property
306 def asn(self) -> int | None:
307 """
308 Search the parsed barcodes for any ASNs.
309 The first barcode that starts with barcode_asn_prefix
310 is considered the ASN to be used.
311 Returns the detected ASN (or None)
312 """
313 asn = None
315 # Ensure the barcodes have been read
316 self.detect()
318 # get the first barcode that starts with barcode_asn_prefix
319 asn_text: str | None = next(
320 (x.value for x in self.barcodes if x.is_asn),
321 None,
322 )
324 if asn_text:
325 logger.debug(f"Found ASN Barcode: {asn_text}")
326 # remove the prefix and remove whitespace
327 asn_text = asn_text[len(self.settings.barcode_asn_prefix) :].strip()
329 # remove non-numeric parts of the remaining string
330 asn_text = re.sub(r"\D", "", asn_text)
332 # now, try parsing the ASN number
333 try:
334 asn = int(asn_text)
335 except ValueError as e:
336 logger.warning(f"Failed to parse ASN number because: {e}")
338 return asn
340 @property
341 def tags(self) -> list[int]:
342 """
343 Search the parsed barcodes for any tags.
344 Returns the detected tag ids (or empty list)
345 """
346 tags: list[int] = []
348 # Ensure the barcodes have been read
349 self.detect()
351 for x in self.barcodes:
352 tag_texts: str = x.value
354 for raw in tag_texts.split(","):
355 try:
356 tag_str: str | None = None
357 for pattern in self.settings.barcode_tag_mapping:
358 if safe_regex_match(pattern, raw, flags=regex_mod.IGNORECASE):
359 sub = self.settings.barcode_tag_mapping[pattern]
360 tag_str = (
361 safe_regex_sub(
362 pattern,
363 sub,
364 raw,
365 flags=regex_mod.IGNORECASE,
366 )
367 if sub
368 else raw
369 )
370 break
372 if tag_str:
373 tag, _ = Tag.objects.get_or_create(
374 name__iexact=tag_str,
375 defaults={"name": tag_str},
376 )
378 logger.debug(
379 f"Found Tag Barcode '{raw}', substituted "
380 f"to '{tag}' and mapped to "
381 f"tag #{tag.pk}.",
382 )
383 tags.append(tag.pk)
385 except Exception as e:
386 logger.error(
387 f"Failed to find or create TAG '{raw}' because: {e}",
388 )
390 return tags
392 def get_separation_pages(self) -> dict[int, bool]:
393 """
394 Search the parsed barcodes for separators and returns a dict of page
395 numbers, which separate the file into new files, together with the
396 information whether to keep the page.
397 """
398 # filter all barcodes for the separator string
399 # get the page numbers of the separating barcodes
400 retain = self.settings.barcode_retain_split_pages
401 separator_pages = {
402 bc.page: retain
403 for bc in self.barcodes
404 if bc.is_separator and (not retain or (retain and bc.page > 0))
405 } # as below, dont include the first page if retain is enabled
407 # add the page numbers of the ASN barcodes
408 # (except for first page, that might lead to infinite loops).
409 if self.settings.barcode_enable_asn:
410 separator_pages = {
411 **separator_pages,
412 **{bc.page: True for bc in self.barcodes if bc.is_asn and bc.page != 0},
413 }
415 # add the page numbers of the TAG barcodes if splitting is enabled
416 # (except for first page, that might lead to infinite loops).
417 if self.settings.barcode_tag_split and self.settings.barcode_enable_tag:
418 separator_pages = {
419 **separator_pages,
420 **{bc.page: True for bc in self.barcodes if bc.is_tag and bc.page != 0},
421 }
423 return separator_pages
425 def separate_pages(self, pages_to_split_on: dict[int, bool]) -> list[Path]:
426 """
427 Separate the provided pdf file on the pages_to_split_on.
428 The pages which are defined by the keys in page_numbers
429 will be removed if the corresponding value is false.
430 Returns a list of (temporary) filepaths to consume.
431 These will need to be deleted later.
432 """
434 document_paths = []
435 fname: str = self.input_doc.original_file.stem
436 with Pdf.open(self.pdf_file) as input_pdf:
437 # Start with an empty document
438 current_document: list[Page] = []
439 # A list of documents, ie a list of lists of pages
440 documents: list[list[Page]] = [current_document]
442 for idx, page in enumerate(input_pdf.pages):
443 # Keep building the new PDF as long as it is not a
444 # separator index
445 if idx not in pages_to_split_on:
446 current_document.append(page)
447 continue
449 # This is a split index
450 # Start a new destination page listing
451 logger.debug(f"Starting new document at idx {idx}")
452 current_document = []
453 documents.append(current_document)
454 keep_page: bool = pages_to_split_on[idx]
455 if keep_page:
456 # Keep the page
457 # (new document is started by asn barcode)
458 current_document.append(page)
460 documents = [x for x in documents if len(x)]
462 logger.debug(f"Split into {len(documents)} new documents")
464 # Write the new documents out
465 for doc_idx, document in enumerate(documents):
466 dst = Pdf.new()
467 dst.pages.extend(document)
469 output_filename = f"{fname}_document_{doc_idx}.pdf"
471 logger.debug(f"pdf no:{doc_idx} has {len(dst.pages)} pages")
472 savepath = Path(self.temp_dir.name) / output_filename
473 with savepath.open("wb") as out:
474 dst.save(out)
476 copy_basic_file_stats(self.input_doc.original_file, savepath)
478 document_paths.append(savepath)
480 return document_paths
483def scannable_mime_types(settings: BarcodeConfig) -> set[str]:
484 """
485 The file types the barcode scan supports with the current settings
486 """
487 if settings.barcode_enable_tiff_support:
488 return {"application/pdf", "image/tiff"}
489 return {"application/pdf"}
492def read_barcodes_zxing(image: Image.Image) -> list[tuple[str, str]]:
493 """
494 Returns the text and format (zxing enum name) of each barcode found in
495 the image
496 """
497 barcodes = []
499 import zxingcpp
501 detected_barcodes = zxingcpp.read_barcodes(image)
502 for barcode in detected_barcodes:
503 if barcode.text:
504 barcodes.append((barcode.text, barcode.format.name))
505 logger.debug(
506 f"Barcode of type {barcode.format} found: {barcode.text}",
507 )
509 return barcodes
512def scan_pdf(pdf_path: Path, settings: BarcodeConfig, work_dir: Path) -> list[Barcode]:
513 """
514 Scans the pages of a PDF as images for barcodes. Errors are not caught,
515 so callers can tell a failed scan from one that found nothing.
516 """
517 barcodes: list[Barcode] = []
519 with Pdf.open(pdf_path) as pdf:
520 num_of_pages = len(pdf.pages)
521 logger.debug(f"PDF has {num_of_pages} pages")
523 # Get limit from configuration
524 barcode_max_pages: int = (
525 num_of_pages if settings.barcode_max_pages == 0 else settings.barcode_max_pages
526 )
528 if barcode_max_pages < num_of_pages: # pragma: no cover
529 logger.debug(
530 f"Barcodes detection will be limited to the first {barcode_max_pages} pages",
531 )
533 for current_page_number in range(min(num_of_pages, barcode_max_pages)):
534 logger.debug(f"Processing page {current_page_number}")
536 # Convert page to image
537 page = convert_from_path(
538 pdf_path,
539 dpi=settings.barcode_dpi,
540 output_folder=work_dir,
541 first_page=current_page_number + 1,
542 last_page=current_page_number + 1,
543 )[0]
545 # Remember filename, since it is lost by upscaling
546 page_filepath = Path(page.filename)
547 logger.debug(f"Image is at {page_filepath}")
549 # Upscale image if configured
550 factor = settings.barcode_upscale
551 if factor > 1.0:
552 logger.debug(
553 f"Upscaling image by {factor} for better barcode detection",
554 )
555 x, y = page.size
556 page = page.resize(
557 (round(x * factor), (round(y * factor))),
558 )
560 for barcode_value, barcode_format in read_barcodes_zxing(page):
561 barcodes.append(
562 Barcode(current_page_number, barcode_value, settings, barcode_format),
563 )
565 # Delete temporary image file
566 page_filepath.unlink()
568 return barcodes
571def read_barcode_values(
572 path: Path,
573 mime_type: str,
574 settings: BarcodeConfig,
575 work_dir: Path,
576) -> list[StoredBarcode] | None:
577 """
578 Reads the barcodes of a file outside of the consumption plugins: for new
579 versions, which skip the barcode plugin, and when reprocessing.
581 Returns None if the file can't be scanned with the current settings.
582 Errors while scanning are raised.
583 """
584 if mime_type not in scannable_mime_types(settings):
585 return None
586 if mime_type == "image/tiff":
587 path = convert_from_tiff_to_pdf(path, work_dir)
588 return [x.stored() for x in scan_pdf(path, settings, work_dir)]