Coverage for paperless/parsers/text.py: 55%
76 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1"""
2Built-in plain-text document parser.
4Handles text/plain, text/csv, and application/csv MIME types by reading the
5file content directly. Thumbnails are generated by rendering a page-sized
6WebP image from the first 100,000 characters using Pillow.
7"""
9from __future__ import annotations
11import logging
12import shutil
13import tempfile
14from pathlib import Path
15from typing import TYPE_CHECKING
16from typing import Self
18from django.conf import settings
19from PIL import Image
20from PIL import ImageDraw
21from PIL import ImageFont
23from paperless.parsers.utils import read_file_handle_unicode_errors
24from paperless.version import __full_version_str__
26if TYPE_CHECKING: 26 ↛ 27line 26 didn't jump to line 27 because the condition on line 26 was never true
27 import datetime
28 from types import TracebackType
30 from paperless.parsers import MetadataEntry
31 from paperless.parsers import ParserContext
33logger = logging.getLogger("paperless.parsing.text")
35_SUPPORTED_MIME_TYPES: dict[str, str] = {
36 "text/plain": ".txt",
37 "text/csv": ".csv",
38 "application/csv": ".csv",
39}
42class TextDocumentParser:
43 """Parse plain-text documents (txt, csv) for Paperless-ngx.
45 This parser reads the file content directly as UTF-8 text and renders a
46 simple thumbnail using Pillow. It does not perform OCR and does not
47 produce a searchable PDF archive copy.
49 Class attributes
50 ----------------
51 name : str
52 Human-readable parser name.
53 version : str
54 Semantic version string, kept in sync with Paperless-ngx releases.
55 author : str
56 Maintainer name.
57 url : str
58 Issue tracker / source URL.
59 """
61 name: str = "Paperless-ngx Text Parser"
62 version: str = __full_version_str__
63 author: str = "Paperless-ngx Contributors"
64 url: str = "https://github.com/paperless-ngx/paperless-ngx"
66 # ------------------------------------------------------------------
67 # Class methods
68 # ------------------------------------------------------------------
70 @classmethod
71 def supported_mime_types(cls) -> dict[str, str]:
72 """Return the MIME types this parser handles.
74 Returns
75 -------
76 dict[str, str]
77 Mapping of MIME type to preferred file extension.
78 """
79 return _SUPPORTED_MIME_TYPES
81 @classmethod
82 def score(
83 cls,
84 mime_type: str,
85 filename: str,
86 path: Path | None = None,
87 ) -> int | None:
88 """Return the priority score for handling this file.
90 Parameters
91 ----------
92 mime_type:
93 Detected MIME type of the file.
94 filename:
95 Original filename including extension.
96 path:
97 Optional filesystem path. Not inspected by this parser.
99 Returns
100 -------
101 int | None
102 10 if the MIME type is supported, otherwise None.
103 """
104 if mime_type in _SUPPORTED_MIME_TYPES: 104 ↛ 106line 104 didn't jump to line 106 because the condition on line 104 was always true
105 return 10
106 return None
108 # ------------------------------------------------------------------
109 # Properties
110 # ------------------------------------------------------------------
112 @property
113 def can_produce_archive(self) -> bool:
114 """Whether this parser can produce a searchable PDF archive copy.
116 Returns
117 -------
118 bool
119 Always False — the text parser does not produce a PDF archive.
120 """
121 return False
123 @property
124 def requires_pdf_rendition(self) -> bool:
125 """Whether the parser must produce a PDF for the frontend to display.
127 Returns
128 -------
129 bool
130 Always False — plain text files are displayable as-is.
131 """
132 return False
134 # ------------------------------------------------------------------
135 # Lifecycle
136 # ------------------------------------------------------------------
138 def __init__(self, logging_group: object = None) -> None:
139 settings.SCRATCH_DIR.mkdir(parents=True, exist_ok=True)
140 self._tempdir = Path(
141 tempfile.mkdtemp(prefix="paperless-", dir=settings.SCRATCH_DIR),
142 )
143 self._text: str | None = None
145 def __enter__(self) -> Self:
146 return self
148 def __exit__(
149 self,
150 exc_type: type[BaseException] | None,
151 exc_val: BaseException | None,
152 exc_tb: TracebackType | None,
153 ) -> None:
154 logger.debug("Cleaning up temporary directory %s", self._tempdir)
155 shutil.rmtree(self._tempdir, ignore_errors=True)
157 # ------------------------------------------------------------------
158 # Core parsing interface
159 # ------------------------------------------------------------------
161 def configure(self, context: ParserContext) -> None:
162 pass
164 def parse(
165 self,
166 document_path: Path,
167 mime_type: str,
168 *,
169 produce_archive: bool = True,
170 ) -> None:
171 """Read the document and store its text content.
173 Parameters
174 ----------
175 document_path:
176 Absolute path to the text file.
177 mime_type:
178 Detected MIME type of the document.
179 produce_archive:
180 Ignored — this parser never produces a PDF archive.
182 Raises
183 ------
184 documents.parsers.ParseError
185 If the file cannot be read.
186 """
187 self._text = read_file_handle_unicode_errors(document_path, log=logger)
189 # ------------------------------------------------------------------
190 # Result accessors
191 # ------------------------------------------------------------------
193 def get_text(self) -> str:
194 """Return the plain-text content extracted during parse.
196 Returns
197 -------
198 str
199 Extracted text, or an empty string if no text could be found.
200 """
201 return self._text or ""
203 def get_date(self) -> datetime.datetime | None:
204 """Return the document date detected during parse.
206 Returns
207 -------
208 datetime.datetime | None
209 Always None — the text parser does not detect dates.
210 """
211 return None
213 def get_archive_path(self) -> Path | None:
214 """Return the path to a generated archive PDF, or None.
216 Returns
217 -------
218 Path | None
219 Always None — the text parser does not produce a PDF archive.
220 """
221 return None
223 # ------------------------------------------------------------------
224 # Thumbnail and metadata
225 # ------------------------------------------------------------------
227 def get_thumbnail(self, document_path: Path, mime_type: str) -> Path:
228 """Render the first portion of the document as a WebP thumbnail.
230 Parameters
231 ----------
232 document_path:
233 Absolute path to the source document.
234 mime_type:
235 Detected MIME type of the document.
237 Returns
238 -------
239 Path
240 Path to the generated WebP thumbnail inside the temporary directory.
241 """
242 max_chars = 100_000
243 file_size_limit = 50 * 1024 * 1024
245 if document_path.stat().st_size > file_size_limit:
246 text = "[File too large to preview]"
247 else:
248 with Path(document_path).open("r", encoding="utf-8", errors="replace") as f:
249 text = f.read(max_chars)
251 img = Image.new("RGB", (500, 700), color="white")
252 draw = ImageDraw.Draw(img)
253 font = ImageFont.truetype(
254 font=settings.THUMBNAIL_FONT_NAME,
255 size=20,
256 layout_engine=ImageFont.Layout.BASIC,
257 )
258 draw.multiline_text((5, 5), text, font=font, fill="black", spacing=4)
260 out_path = self._tempdir / "thumb.webp"
261 img.save(out_path, format="WEBP")
263 return out_path
265 def get_page_count(
266 self,
267 document_path: Path,
268 mime_type: str,
269 ) -> int | None:
270 """Return the number of pages in the document.
272 Parameters
273 ----------
274 document_path:
275 Absolute path to the source document.
276 mime_type:
277 Detected MIME type of the document.
279 Returns
280 -------
281 int | None
282 Always None — page count is not meaningful for plain text.
283 """
284 return None
286 def extract_metadata(
287 self,
288 document_path: Path,
289 mime_type: str,
290 ) -> list[MetadataEntry]:
291 """Extract format-specific metadata from the document.
293 Returns
294 -------
295 list[MetadataEntry]
296 Always ``[]`` — plain text files carry no structured metadata.
297 """
298 return []