Coverage for documents/parsers.py: 32%
134 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import logging
4import mimetypes
5import os
6import shutil
7import subprocess
8import tempfile
9from pathlib import Path
10from typing import TYPE_CHECKING
12from django.conf import settings
14from documents.loggers import LoggingMixin
15from documents.utils import copy_file_with_basic_stats
16from documents.utils import run_subprocess
17from paperless.parsers.registry import get_parser_registry
19if TYPE_CHECKING: 19 ↛ 20line 19 didn't jump to line 20 because the condition on line 19 was never true
20 import datetime
22logger = logging.getLogger("paperless.parsing")
25def is_mime_type_supported(mime_type: str) -> bool:
26 """
27 Returns True if the mime type is supported, False otherwise
28 """
29 return get_parser_registry().get_parser_for_file(mime_type, "") is not None
32def get_default_file_extension(mime_type: str) -> str:
33 """
34 Returns the default file extension for a mimetype, or
35 an empty string if it could not be determined
36 """
37 parser_class = get_parser_registry().get_parser_for_file(mime_type, "")
38 if parser_class is not None: 38 ↛ 43line 38 didn't jump to line 43 because the condition on line 38 was always true
39 supported = parser_class.supported_mime_types()
40 if mime_type in supported: 40 ↛ 43line 40 didn't jump to line 43 because the condition on line 40 was always true
41 return supported[mime_type]
43 ext = mimetypes.guess_extension(mime_type)
44 return ext if ext else ""
47def is_file_ext_supported(ext: str) -> bool:
48 """
49 Returns True if the file extension is supported, False otherwise
50 TODO: Investigate why this really exists, why not use mimetype
51 """
52 if ext:
53 return ext.lower() in get_supported_file_extensions()
54 else:
55 return False
58def get_supported_file_extensions() -> set[str]:
59 extensions = set()
60 for parser_class in get_parser_registry().all_parsers():
61 for mime_type, ext in parser_class.supported_mime_types().items():
62 extensions.update(mimetypes.guess_all_extensions(mime_type))
63 # Python's stdlib might be behind, so also add what the parser
64 # says is the default extension
65 # This makes image/webp supported on Python < 3.11
66 extensions.add(ext)
68 return extensions
71def run_convert(
72 input_file,
73 output_file,
74 *,
75 density=None,
76 scale=None,
77 alpha=None,
78 strip=False,
79 trim=False,
80 type=None,
81 depth=None,
82 auto_orient=False,
83 use_cropbox=False,
84 extra=None,
85 logging_group=None,
86) -> None:
87 environment = os.environ.copy()
88 if settings.CONVERT_MEMORY_LIMIT:
89 # MAGICK_MEMORY_LIMIT sets the maximum amount of RAM the pixel cache can use.
90 # MAGICK_MAP_LIMIT sets the maximum amount of memory-mapped I/O allowed.
91 #
92 # For large-format documents ImageMagick will hit the RAM limit and
93 # immediately try to "map" the remaining data. If MAGICK_MAP_LIMIT isn't
94 # also set, the process may trigger an OOM kill because the default
95 # system/policy map limit is often too restrictive for these massive bitmaps.
96 environment["MAGICK_MEMORY_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
97 environment["MAGICK_MAP_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
98 if settings.CONVERT_TMPDIR:
99 environment["MAGICK_TMPDIR"] = settings.CONVERT_TMPDIR
101 args = [settings.CONVERT_BINARY]
102 args += ["-density", str(density)] if density else []
103 args += ["-scale", str(scale)] if scale else []
104 args += ["-alpha", str(alpha)] if alpha else []
105 args += ["-strip"] if strip else []
106 args += ["-trim"] if trim else []
107 args += ["-type", str(type)] if type else []
108 args += ["-depth", str(depth)] if depth else []
109 args += ["-auto-orient"] if auto_orient else []
110 args += ["-define", "pdf:use-cropbox=true"] if use_cropbox else []
111 args += [str(input_file), str(output_file)]
113 logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
115 try:
116 run_subprocess(args, environment, logger)
117 except subprocess.CalledProcessError as e:
118 raise ParseError(f"Convert failed at {args}") from e
119 except Exception as e: # pragma: no cover
120 raise ParseError("Unknown error running convert") from e
123def get_default_thumbnail() -> Path:
124 """
125 Returns the path to a generic thumbnail
126 """
127 return (Path(__file__).parent / "resources" / "document.webp").resolve()
130def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path:
131 out_path: Path = Path(temp_dir) / "convert_gs.webp"
133 # if convert fails, fall back to extracting
134 # the first PDF page as a PNG using Ghostscript
135 logger.warning(
136 "Thumbnail generation with ImageMagick failed, falling back "
137 "to ghostscript. Check your /etc/ImageMagick-x/policy.xml!",
138 extra={"group": logging_group},
139 )
140 # Ghostscript doesn't handle WebP outputs
141 gs_out_path: Path = Path(temp_dir) / "gs_out.png"
142 cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path]
144 try:
145 try:
146 run_subprocess(cmd, logger=logger)
147 except subprocess.CalledProcessError as e:
148 raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e
149 # then run convert on the output from gs to make WebP
150 run_convert(
151 density=300,
152 scale="500x5000>",
153 alpha="remove",
154 strip=True,
155 trim=False,
156 auto_orient=True,
157 input_file=gs_out_path,
158 output_file=out_path,
159 logging_group=logging_group,
160 )
162 return out_path
164 except ParseError as e:
165 logger.error(f"Unable to make thumbnail with Ghostscript: {e}")
166 # The caller might expect a generated thumbnail that can be moved,
167 # so we need to copy it before it gets moved.
168 # https://github.com/paperless-ngx/paperless-ngx/issues/3631
169 default_thumbnail_path: Path = Path(temp_dir) / "document.webp"
170 copy_file_with_basic_stats(get_default_thumbnail(), default_thumbnail_path)
171 return default_thumbnail_path
174def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) -> Path:
175 """
176 The thumbnail of a PDF is just a 500px wide image of the first page.
177 """
178 out_path: Path = temp_dir / "convert.webp"
180 # Run convert to get a decent thumbnail
181 try:
182 run_convert(
183 density=300,
184 scale="500x5000>",
185 alpha="remove",
186 strip=True,
187 trim=False,
188 auto_orient=True,
189 use_cropbox=True,
190 input_file=f"{in_path}[0]",
191 output_file=str(out_path),
192 logging_group=logging_group,
193 )
194 except ParseError as e:
195 logger.error(f"Unable to make thumbnail with convert: {e}")
196 out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group)
198 return out_path
201class ParseError(Exception):
202 pass
205class DocumentParser(LoggingMixin):
206 """
207 Subclass this to make your own parser. Have a look at
208 `paperless_tesseract.parsers` for inspiration.
209 """
211 logging_name = "paperless.parsing"
213 def __init__(self, logging_group, progress_callback=None) -> None:
214 super().__init__()
215 self.renew_logging_group()
216 self.logging_group = logging_group
217 self.settings = self.get_settings()
218 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True)
219 self.tempdir = Path(
220 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR),
221 )
223 self.archive_path = None
224 self.text = None
225 self.date: datetime.datetime | None = None
226 self.progress_callback = progress_callback
228 def progress(self, current_progress, max_progress) -> None:
229 if self.progress_callback:
230 self.progress_callback(current_progress, max_progress)
232 def get_settings(self): # pragma: no cover
233 """
234 A parser must implement this
235 """
236 raise NotImplementedError
238 def read_file_handle_unicode_errors(self, filepath: Path) -> str:
239 """
240 Helper utility for reading from a file, and handling a problem with its
241 unicode, falling back to ignoring the error to remove the invalid bytes
242 """
243 try:
244 text = filepath.read_text(encoding="utf-8")
245 except UnicodeDecodeError as e:
246 self.log.warning(f"Unicode error during text reading, continuing: {e}")
247 text = filepath.read_bytes().decode("utf-8", errors="replace")
248 return text
250 def extract_metadata(self, document_path, mime_type):
251 return []
253 def get_page_count(self, document_path, mime_type) -> None:
254 return None
256 def parse(self, document_path, mime_type, file_name=None):
257 raise NotImplementedError
259 def get_archive_path(self):
260 return self.archive_path
262 def get_thumbnail(self, document_path, mime_type, file_name=None):
263 """
264 Returns the path to a file we can use as a thumbnail for this document.
265 """
266 raise NotImplementedError
268 def get_text(self):
269 return self.text
271 def get_date(self) -> datetime.datetime | None:
272 return self.date
274 def cleanup(self) -> None:
275 self.log.debug(f"Deleting directory {self.tempdir}")
276 shutil.rmtree(self.tempdir)