Coverage for documents/parsers.py: 32%

134 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import logging 

4import mimetypes 

5import os 

6import shutil 

7import subprocess 

8import tempfile 

9from pathlib import Path 

10from typing import TYPE_CHECKING 

11 

12from django.conf import settings 

13 

14from documents.loggers import LoggingMixin 

15from documents.utils import copy_file_with_basic_stats 

16from documents.utils import run_subprocess 

17from paperless.parsers.registry import get_parser_registry 

18 

19if TYPE_CHECKING: 19 ↛ 20line 19 didn't jump to line 20 because the condition on line 19 was never true

20 import datetime 

21 

22logger = logging.getLogger("paperless.parsing") 

23 

24 

25def is_mime_type_supported(mime_type: str) -> bool: 

26 """ 

27 Returns True if the mime type is supported, False otherwise 

28 """ 

29 return get_parser_registry().get_parser_for_file(mime_type, "") is not None 

30 

31 

32def get_default_file_extension(mime_type: str) -> str: 

33 """ 

34 Returns the default file extension for a mimetype, or 

35 an empty string if it could not be determined 

36 """ 

37 parser_class = get_parser_registry().get_parser_for_file(mime_type, "") 

38 if parser_class is not None: 38 ↛ 43line 38 didn't jump to line 43 because the condition on line 38 was always true

39 supported = parser_class.supported_mime_types() 

40 if mime_type in supported: 40 ↛ 43line 40 didn't jump to line 43 because the condition on line 40 was always true

41 return supported[mime_type] 

42 

43 ext = mimetypes.guess_extension(mime_type) 

44 return ext if ext else "" 

45 

46 

47def is_file_ext_supported(ext: str) -> bool: 

48 """ 

49 Returns True if the file extension is supported, False otherwise 

50 TODO: Investigate why this really exists, why not use mimetype 

51 """ 

52 if ext: 

53 return ext.lower() in get_supported_file_extensions() 

54 else: 

55 return False 

56 

57 

58def get_supported_file_extensions() -> set[str]: 

59 extensions = set() 

60 for parser_class in get_parser_registry().all_parsers(): 

61 for mime_type, ext in parser_class.supported_mime_types().items(): 

62 extensions.update(mimetypes.guess_all_extensions(mime_type)) 

63 # Python's stdlib might be behind, so also add what the parser 

64 # says is the default extension 

65 # This makes image/webp supported on Python < 3.11 

66 extensions.add(ext) 

67 

68 return extensions 

69 

70 

71def run_convert( 

72 input_file, 

73 output_file, 

74 *, 

75 density=None, 

76 scale=None, 

77 alpha=None, 

78 strip=False, 

79 trim=False, 

80 type=None, 

81 depth=None, 

82 auto_orient=False, 

83 use_cropbox=False, 

84 extra=None, 

85 logging_group=None, 

86) -> None: 

87 environment = os.environ.copy() 

88 if settings.CONVERT_MEMORY_LIMIT: 

89 # MAGICK_MEMORY_LIMIT sets the maximum amount of RAM the pixel cache can use. 

90 # MAGICK_MAP_LIMIT sets the maximum amount of memory-mapped I/O allowed. 

91 # 

92 # For large-format documents ImageMagick will hit the RAM limit and 

93 # immediately try to "map" the remaining data. If MAGICK_MAP_LIMIT isn't 

94 # also set, the process may trigger an OOM kill because the default 

95 # system/policy map limit is often too restrictive for these massive bitmaps. 

96 environment["MAGICK_MEMORY_LIMIT"] = settings.CONVERT_MEMORY_LIMIT 

97 environment["MAGICK_MAP_LIMIT"] = settings.CONVERT_MEMORY_LIMIT 

98 if settings.CONVERT_TMPDIR: 

99 environment["MAGICK_TMPDIR"] = settings.CONVERT_TMPDIR 

100 

101 args = [settings.CONVERT_BINARY] 

102 args += ["-density", str(density)] if density else [] 

103 args += ["-scale", str(scale)] if scale else [] 

104 args += ["-alpha", str(alpha)] if alpha else [] 

105 args += ["-strip"] if strip else [] 

106 args += ["-trim"] if trim else [] 

107 args += ["-type", str(type)] if type else [] 

108 args += ["-depth", str(depth)] if depth else [] 

109 args += ["-auto-orient"] if auto_orient else [] 

110 args += ["-define", "pdf:use-cropbox=true"] if use_cropbox else [] 

111 args += [str(input_file), str(output_file)] 

112 

113 logger.debug("Execute: " + " ".join(args), extra={"group": logging_group}) 

114 

115 try: 

116 run_subprocess(args, environment, logger) 

117 except subprocess.CalledProcessError as e: 

118 raise ParseError(f"Convert failed at {args}") from e 

119 except Exception as e: # pragma: no cover 

120 raise ParseError("Unknown error running convert") from e 

121 

122 

123def get_default_thumbnail() -> Path: 

124 """ 

125 Returns the path to a generic thumbnail 

126 """ 

127 return (Path(__file__).parent / "resources" / "document.webp").resolve() 

128 

129 

130def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path: 

131 out_path: Path = Path(temp_dir) / "convert_gs.webp" 

132 

133 # if convert fails, fall back to extracting 

134 # the first PDF page as a PNG using Ghostscript 

135 logger.warning( 

136 "Thumbnail generation with ImageMagick failed, falling back " 

137 "to ghostscript. Check your /etc/ImageMagick-x/policy.xml!", 

138 extra={"group": logging_group}, 

139 ) 

140 # Ghostscript doesn't handle WebP outputs 

141 gs_out_path: Path = Path(temp_dir) / "gs_out.png" 

142 cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path] 

143 

144 try: 

145 try: 

146 run_subprocess(cmd, logger=logger) 

147 except subprocess.CalledProcessError as e: 

148 raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e 

149 # then run convert on the output from gs to make WebP 

150 run_convert( 

151 density=300, 

152 scale="500x5000>", 

153 alpha="remove", 

154 strip=True, 

155 trim=False, 

156 auto_orient=True, 

157 input_file=gs_out_path, 

158 output_file=out_path, 

159 logging_group=logging_group, 

160 ) 

161 

162 return out_path 

163 

164 except ParseError as e: 

165 logger.error(f"Unable to make thumbnail with Ghostscript: {e}") 

166 # The caller might expect a generated thumbnail that can be moved, 

167 # so we need to copy it before it gets moved. 

168 # https://github.com/paperless-ngx/paperless-ngx/issues/3631 

169 default_thumbnail_path: Path = Path(temp_dir) / "document.webp" 

170 copy_file_with_basic_stats(get_default_thumbnail(), default_thumbnail_path) 

171 return default_thumbnail_path 

172 

173 

174def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) -> Path: 

175 """ 

176 The thumbnail of a PDF is just a 500px wide image of the first page. 

177 """ 

178 out_path: Path = temp_dir / "convert.webp" 

179 

180 # Run convert to get a decent thumbnail 

181 try: 

182 run_convert( 

183 density=300, 

184 scale="500x5000>", 

185 alpha="remove", 

186 strip=True, 

187 trim=False, 

188 auto_orient=True, 

189 use_cropbox=True, 

190 input_file=f"{in_path}[0]", 

191 output_file=str(out_path), 

192 logging_group=logging_group, 

193 ) 

194 except ParseError as e: 

195 logger.error(f"Unable to make thumbnail with convert: {e}") 

196 out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group) 

197 

198 return out_path 

199 

200 

201class ParseError(Exception): 

202 pass 

203 

204 

205class DocumentParser(LoggingMixin): 

206 """ 

207 Subclass this to make your own parser. Have a look at 

208 `paperless_tesseract.parsers` for inspiration. 

209 """ 

210 

211 logging_name = "paperless.parsing" 

212 

213 def __init__(self, logging_group, progress_callback=None) -> None: 

214 super().__init__() 

215 self.renew_logging_group() 

216 self.logging_group = logging_group 

217 self.settings = self.get_settings() 

218 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True) 

219 self.tempdir = Path( 

220 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR), 

221 ) 

222 

223 self.archive_path = None 

224 self.text = None 

225 self.date: datetime.datetime | None = None 

226 self.progress_callback = progress_callback 

227 

228 def progress(self, current_progress, max_progress) -> None: 

229 if self.progress_callback: 

230 self.progress_callback(current_progress, max_progress) 

231 

232 def get_settings(self): # pragma: no cover 

233 """ 

234 A parser must implement this 

235 """ 

236 raise NotImplementedError 

237 

238 def read_file_handle_unicode_errors(self, filepath: Path) -> str: 

239 """ 

240 Helper utility for reading from a file, and handling a problem with its 

241 unicode, falling back to ignoring the error to remove the invalid bytes 

242 """ 

243 try: 

244 text = filepath.read_text(encoding="utf-8") 

245 except UnicodeDecodeError as e: 

246 self.log.warning(f"Unicode error during text reading, continuing: {e}") 

247 text = filepath.read_bytes().decode("utf-8", errors="replace") 

248 return text 

249 

250 def extract_metadata(self, document_path, mime_type): 

251 return [] 

252 

253 def get_page_count(self, document_path, mime_type) -> None: 

254 return None 

255 

256 def parse(self, document_path, mime_type, file_name=None): 

257 raise NotImplementedError 

258 

259 def get_archive_path(self): 

260 return self.archive_path 

261 

262 def get_thumbnail(self, document_path, mime_type, file_name=None): 

263 """ 

264 Returns the path to a file we can use as a thumbnail for this document. 

265 """ 

266 raise NotImplementedError 

267 

268 def get_text(self): 

269 return self.text 

270 

271 def get_date(self) -> datetime.datetime | None: 

272 return self.date 

273 

274 def cleanup(self) -> None: 

275 self.log.debug(f"Deleting directory {self.tempdir}") 

276 shutil.rmtree(self.tempdir)