Coverage for paperless/parsers/text.py: 55%

76 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1""" 

2Built-in plain-text document parser. 

3 

4Handles text/plain, text/csv, and application/csv MIME types by reading the 

5file content directly. Thumbnails are generated by rendering a page-sized 

6WebP image from the first 100,000 characters using Pillow. 

7""" 

8 

9from __future__ import annotations 

10 

11import logging 

12import shutil 

13import tempfile 

14from pathlib import Path 

15from typing import TYPE_CHECKING 

16from typing import Self 

17 

18from django.conf import settings 

19from PIL import Image 

20from PIL import ImageDraw 

21from PIL import ImageFont 

22 

23from paperless.parsers.utils import read_file_handle_unicode_errors 

24from paperless.version import __full_version_str__ 

25 

26if TYPE_CHECKING: 26 ↛ 27line 26 didn't jump to line 27 because the condition on line 26 was never true

27 import datetime 

28 from types import TracebackType 

29 

30 from paperless.parsers import MetadataEntry 

31 from paperless.parsers import ParserContext 

32 

33logger = logging.getLogger("paperless.parsing.text") 

34 

35_SUPPORTED_MIME_TYPES: dict[str, str] = { 

36 "text/plain": ".txt", 

37 "text/csv": ".csv", 

38 "application/csv": ".csv", 

39} 

40 

41 

42class TextDocumentParser: 

43 """Parse plain-text documents (txt, csv) for Paperless-ngx. 

44 

45 This parser reads the file content directly as UTF-8 text and renders a 

46 simple thumbnail using Pillow. It does not perform OCR and does not 

47 produce a searchable PDF archive copy. 

48 

49 Class attributes 

50 ---------------- 

51 name : str 

52 Human-readable parser name. 

53 version : str 

54 Semantic version string, kept in sync with Paperless-ngx releases. 

55 author : str 

56 Maintainer name. 

57 url : str 

58 Issue tracker / source URL. 

59 """ 

60 

61 name: str = "Paperless-ngx Text Parser" 

62 version: str = __full_version_str__ 

63 author: str = "Paperless-ngx Contributors" 

64 url: str = "https://github.com/paperless-ngx/paperless-ngx" 

65 

66 # ------------------------------------------------------------------ 

67 # Class methods 

68 # ------------------------------------------------------------------ 

69 

70 @classmethod 

71 def supported_mime_types(cls) -> dict[str, str]: 

72 """Return the MIME types this parser handles. 

73 

74 Returns 

75 ------- 

76 dict[str, str] 

77 Mapping of MIME type to preferred file extension. 

78 """ 

79 return _SUPPORTED_MIME_TYPES 

80 

81 @classmethod 

82 def score( 

83 cls, 

84 mime_type: str, 

85 filename: str, 

86 path: Path | None = None, 

87 ) -> int | None: 

88 """Return the priority score for handling this file. 

89 

90 Parameters 

91 ---------- 

92 mime_type: 

93 Detected MIME type of the file. 

94 filename: 

95 Original filename including extension. 

96 path: 

97 Optional filesystem path. Not inspected by this parser. 

98 

99 Returns 

100 ------- 

101 int | None 

102 10 if the MIME type is supported, otherwise None. 

103 """ 

104 if mime_type in _SUPPORTED_MIME_TYPES: 104 ↛ 106line 104 didn't jump to line 106 because the condition on line 104 was always true

105 return 10 

106 return None 

107 

108 # ------------------------------------------------------------------ 

109 # Properties 

110 # ------------------------------------------------------------------ 

111 

112 @property 

113 def can_produce_archive(self) -> bool: 

114 """Whether this parser can produce a searchable PDF archive copy. 

115 

116 Returns 

117 ------- 

118 bool 

119 Always False — the text parser does not produce a PDF archive. 

120 """ 

121 return False 

122 

123 @property 

124 def requires_pdf_rendition(self) -> bool: 

125 """Whether the parser must produce a PDF for the frontend to display. 

126 

127 Returns 

128 ------- 

129 bool 

130 Always False — plain text files are displayable as-is. 

131 """ 

132 return False 

133 

134 # ------------------------------------------------------------------ 

135 # Lifecycle 

136 # ------------------------------------------------------------------ 

137 

138 def __init__(self, logging_group: object = None) -> None: 

139 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True) 

140 self._tempdir = Path( 

141 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR), 

142 ) 

143 self._text: str | None = None 

144 

145 def __enter__(self) -> Self: 

146 return self 

147 

148 def __exit__( 

149 self, 

150 exc_type: type[BaseException] | None, 

151 exc_val: BaseException | None, 

152 exc_tb: TracebackType | None, 

153 ) -> None: 

154 logger.debug("Cleaning up temporary directory %s", self._tempdir) 

155 shutil.rmtree(self._tempdir, ignore_errors=True) 

156 

157 # ------------------------------------------------------------------ 

158 # Core parsing interface 

159 # ------------------------------------------------------------------ 

160 

161 def configure(self, context: ParserContext) -> None: 

162 pass 

163 

164 def parse( 

165 self, 

166 document_path: Path, 

167 mime_type: str, 

168 *, 

169 produce_archive: bool = True, 

170 ) -> None: 

171 """Read the document and store its text content. 

172 

173 Parameters 

174 ---------- 

175 document_path: 

176 Absolute path to the text file. 

177 mime_type: 

178 Detected MIME type of the document. 

179 produce_archive: 

180 Ignored — this parser never produces a PDF archive. 

181 

182 Raises 

183 ------ 

184 documents.parsers.ParseError 

185 If the file cannot be read. 

186 """ 

187 self._text = read_file_handle_unicode_errors(document_path, log=logger) 

188 

189 # ------------------------------------------------------------------ 

190 # Result accessors 

191 # ------------------------------------------------------------------ 

192 

193 def get_text(self) -> str: 

194 """Return the plain-text content extracted during parse. 

195 

196 Returns 

197 ------- 

198 str 

199 Extracted text, or an empty string if no text could be found. 

200 """ 

201 return self._text or "" 

202 

203 def get_date(self) -> datetime.datetime | None: 

204 """Return the document date detected during parse. 

205 

206 Returns 

207 ------- 

208 datetime.datetime | None 

209 Always None — the text parser does not detect dates. 

210 """ 

211 return None 

212 

213 def get_archive_path(self) -> Path | None: 

214 """Return the path to a generated archive PDF, or None. 

215 

216 Returns 

217 ------- 

218 Path | None 

219 Always None — the text parser does not produce a PDF archive. 

220 """ 

221 return None 

222 

223 # ------------------------------------------------------------------ 

224 # Thumbnail and metadata 

225 # ------------------------------------------------------------------ 

226 

227 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path: 

228 """Render the first portion of the document as a WebP thumbnail. 

229 

230 Parameters 

231 ---------- 

232 document_path: 

233 Absolute path to the source document. 

234 mime_type: 

235 Detected MIME type of the document. 

236 

237 Returns 

238 ------- 

239 Path 

240 Path to the generated WebP thumbnail inside the temporary directory. 

241 """ 

242 max_chars = 100_000 

243 file_size_limit = 50 * 1024 * 1024 

244 

245 if document_path.stat().st_size > file_size_limit: 

246 text = "[File too large to preview]" 

247 else: 

248 with Path(document_path).open("r", encoding="utf-8", errors="replace") as f: 

249 text = f.read(max_chars) 

250 

251 img = Image.new("RGB", (500, 700), color="white") 

252 draw = ImageDraw.Draw(img) 

253 font = ImageFont.truetype( 

254 font=settings.THUMBNAIL_FONT_NAME, 

255 size=20, 

256 layout_engine=ImageFont.Layout.BASIC, 

257 ) 

258 draw.multiline_text((5, 5), text, font=font, fill="black", spacing=4) 

259 

260 out_path = self._tempdir / "thumb.webp" 

261 img.save(out_path, format="WEBP") 

262 

263 return out_path 

264 

265 def get_page_count( 

266 self, 

267 document_path: Path, 

268 mime_type: str, 

269 ) -> int | None: 

270 """Return the number of pages in the document. 

271 

272 Parameters 

273 ---------- 

274 document_path: 

275 Absolute path to the source document. 

276 mime_type: 

277 Detected MIME type of the document. 

278 

279 Returns 

280 ------- 

281 int | None 

282 Always None — page count is not meaningful for plain text. 

283 """ 

284 return None 

285 

286 def extract_metadata( 

287 self, 

288 document_path: Path, 

289 mime_type: str, 

290 ) -> list[MetadataEntry]: 

291 """Extract format-specific metadata from the document. 

292 

293 Returns 

294 ------- 

295 list[MetadataEntry] 

296 Always ``[]`` — plain text files carry no structured metadata. 

297 """ 

298 return []