Coverage for documents/templating/filepath.py: 73%
121 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1import logging
2import os
3import re
4import unicodedata
5from collections.abc import Iterable
6from pathlib import PurePath
8import pathvalidate
9from django.utils import timezone
10from django.utils.text import slugify as django_slugify
11from jinja2 import StrictUndefined
12from jinja2 import Template
13from jinja2 import TemplateSyntaxError
14from jinja2 import UndefinedError
15from jinja2 import make_logging_undefined
16from jinja2.sandbox import SecurityError
18from documents.models import Correspondent
19from documents.models import CustomField
20from documents.models import CustomFieldInstance
21from documents.models import Document
22from documents.models import DocumentType
23from documents.models import StoragePath
24from documents.models import Tag
25from documents.templating.environment import _template_environment
26from documents.templating.filters import format_datetime
27from documents.templating.filters import get_cf_value
28from documents.templating.filters import localize_date
30logger = logging.getLogger("paperless.templating")
32_LogStrictUndefined = make_logging_undefined(logger, StrictUndefined)
35class FilePathTemplate(Template):
36 def render(self, *args, **kwargs) -> str:
37 def clean_filepath(value: str) -> str:
38 """
39 Clean up a filepath by:
40 1. Normalizing Unicode to NFC form to prevent byte-level mismatches
41 2. Removing newlines and carriage returns
42 3. Removing extra spaces before and after forward slashes
43 4. Preserving spaces in other parts of the path
44 """
45 value = unicodedata.normalize("NFC", value)
46 value = value.replace("\n", "").replace("\r", "")
47 value = re.sub(r"\s*/\s*", "/", value)
49 # We remove trailing and leading separators, as these are always relative paths, not absolute, even if the user
50 # tries
51 return value.strip().strip(os.sep)
53 original_render = super().render(*args, **kwargs)
55 return clean_filepath(original_render)
58class PlaceholderString(str):
59 """
60 String subclass used as a sentinel for empty metadata values inside templates.
62 - Renders as \"-none-\" to preserve existing filename cleaning logic.
63 - Compares equal to either \"-none-\" or \"none\" so templates can check for either.
64 - Evaluates to False so {% if correspondent %} behaves intuitively.
65 """
67 def __new__(cls, value: str = "-none-"):
68 return super().__new__(cls, value)
70 def __bool__(self) -> bool:
71 return False
73 def __eq__(self, other) -> bool:
74 if isinstance(other, str) and other == "none":
75 other = "-none-"
76 return super().__eq__(other)
78 def __ne__(self, other) -> bool:
79 return not self.__eq__(other)
82NO_VALUE_PLACEHOLDER = PlaceholderString("-none-")
85class MatchingModelContext:
86 """
87 Safe template context for related objects.
89 Keeps legacy behavior where including the object ina template yields the related object's
90 name as a string, while still exposing limited attributes.
91 """
93 def __init__(self, *, id: int, name: str, path: str | None = None):
94 self.id = id
95 self.name = name
96 self.path = path
98 def __str__(self) -> str:
99 return self.name
102_template_environment.undefined = _LogStrictUndefined
104_template_environment.filters["get_cf_value"] = get_cf_value
106_template_environment.filters["datetime"] = format_datetime
108_template_environment.filters["slugify"] = django_slugify
110_template_environment.filters["localize_date"] = localize_date
113def create_dummy_document():
114 """
115 Create a dummy Document instance with all possible fields filled
116 """
117 # Populate the document with representative values for every field
118 dummy_doc = Document(
119 pk=1,
120 title="Sample Title",
121 correspondent=Correspondent(name="Sample Correspondent"),
122 storage_path=StoragePath(path="/dummy/path"),
123 document_type=DocumentType(name="Sample Type"),
124 content="This is some sample document content.",
125 mime_type="application/pdf",
126 checksum="dummychecksum12345678901234567890123456789012",
127 archive_checksum="dummyarchivechecksum123456789012345678901234",
128 page_count=5,
129 created=timezone.now(),
130 modified=timezone.now(),
131 added=timezone.now(),
132 filename="/dummy/filename.pdf",
133 archive_filename="/dummy/archive_filename.pdf",
134 original_filename="original_file.pdf",
135 archive_serial_number=12345,
136 )
137 return dummy_doc
140def get_creation_date_context(document: Document) -> dict[str, str]:
141 """
142 Given a Document, localizes the creation date and builds a context dictionary with some common, shorthand
143 formatted values from it
144 """
145 return {
146 "created": document.created.isoformat(),
147 "created_year": document.created.strftime("%Y"),
148 "created_year_short": document.created.strftime("%y"),
149 "created_month": document.created.strftime("%m"),
150 "created_month_name": document.created.strftime("%B"),
151 "created_month_name_short": document.created.strftime("%b"),
152 "created_day": document.created.strftime("%d"),
153 }
156def get_added_date_context(document: Document) -> dict[str, str]:
157 """
158 Given a Document, localizes the added date and builds a context dictionary with some common, shorthand
159 formatted values from it
160 """
161 local_added = timezone.localdate(document.added)
163 return {
164 "added": local_added.isoformat(),
165 "added_year": local_added.strftime("%Y"),
166 "added_year_short": local_added.strftime("%y"),
167 "added_month": local_added.strftime("%m"),
168 "added_month_name": local_added.strftime("%B"),
169 "added_month_name_short": local_added.strftime("%b"),
170 "added_day": local_added.strftime("%d"),
171 }
174def get_basic_metadata_context(
175 document: Document,
176 *,
177 no_value_default: str = NO_VALUE_PLACEHOLDER,
178) -> dict[str, str]:
179 """
180 Given a Document, constructs some basic information about it. If certain values are not set,
181 they will be replaced with the no_value_default.
183 Regardless of set or not, the values will be sanitized
184 """
185 return {
186 "title": pathvalidate.sanitize_filename(
187 unicodedata.normalize("NFC", document.title),
188 replacement_text="-",
189 ),
190 "correspondent": pathvalidate.sanitize_filename(
191 unicodedata.normalize("NFC", document.correspondent.name),
192 replacement_text="-",
193 )
194 if document.correspondent
195 else no_value_default,
196 "document_type": pathvalidate.sanitize_filename(
197 unicodedata.normalize("NFC", document.document_type.name),
198 replacement_text="-",
199 )
200 if document.document_type
201 else no_value_default,
202 "asn": str(document.archive_serial_number)
203 if document.archive_serial_number
204 else no_value_default,
205 "owner_username": document.owner.username
206 if document.owner
207 else no_value_default,
208 "original_name": unicodedata.normalize(
209 "NFC",
210 PurePath(document.original_filename).with_suffix("").name,
211 )
212 if document.original_filename
213 else no_value_default,
214 "doc_pk": f"{document.pk:07}",
215 }
218def get_safe_document_context(
219 document: Document,
220 tags: Iterable[Tag],
221) -> dict[str, object]:
222 """
223 Build a document context object to avoid supplying entire model instance.
224 """
225 return {
226 "id": document.pk,
227 "pk": document.pk,
228 "title": document.title,
229 "content": document.content,
230 "page_count": document.page_count,
231 "created": document.created,
232 "added": document.added,
233 "modified": document.modified,
234 "archive_serial_number": document.archive_serial_number,
235 "mime_type": document.mime_type,
236 "checksum": document.checksum,
237 "archive_checksum": document.archive_checksum,
238 "filename": document.filename,
239 "archive_filename": document.archive_filename,
240 "original_filename": document.original_filename,
241 "owner": {"username": document.owner.username, "id": document.owner.id}
242 if document.owner
243 else None,
244 "tags": [{"name": tag.name, "id": tag.id} for tag in tags],
245 "correspondent": (
246 MatchingModelContext(
247 name=document.correspondent.name,
248 id=document.correspondent.id,
249 )
250 if document.correspondent
251 else None
252 ),
253 "document_type": (
254 MatchingModelContext(
255 name=document.document_type.name,
256 id=document.document_type.id,
257 )
258 if document.document_type
259 else None
260 ),
261 "storage_path": MatchingModelContext(
262 name=document.storage_path.name,
263 path=document.storage_path.path,
264 id=document.storage_path.id,
265 )
266 if document.storage_path
267 else None,
268 }
271def get_tags_context(tags: Iterable[Tag]) -> dict[str, str | list[str]]:
272 """
273 Given an Iterable of tags, constructs some context from them for usage
274 """
275 return {
276 "tag_list": pathvalidate.sanitize_filename(
277 ",".join(
278 sorted(unicodedata.normalize("NFC", tag.name) for tag in tags),
279 ),
280 replacement_text="-",
281 ),
282 # Assumed to be ordered, but a template could loop through to find what they want
283 "tag_name_list": [unicodedata.normalize("NFC", x.name) for x in tags],
284 }
287def get_custom_fields_context(
288 custom_fields: Iterable[CustomFieldInstance],
289) -> dict[str, dict[str, dict[str, str]]]:
290 """
291 Given an Iterable of CustomFieldInstance, builds a dictionary mapping the field name
292 to its type and value
293 """
294 field_data = {"custom_fields": {}}
295 for field_instance in custom_fields:
296 type_ = pathvalidate.sanitize_filename(
297 field_instance.field.data_type,
298 replacement_text="-",
299 )
300 if field_instance.value is None: 300 ↛ 301line 300 didn't jump to line 301 because the condition on line 300 was never true
301 value = None
302 # String types need to be sanitized
303 elif field_instance.field.data_type in { 303 ↛ 313line 303 didn't jump to line 313 because the condition on line 303 was always true
304 CustomField.FieldDataType.MONETARY,
305 CustomField.FieldDataType.STRING,
306 CustomField.FieldDataType.URL,
307 CustomField.FieldDataType.LONG_TEXT,
308 }:
309 value = pathvalidate.sanitize_filename(
310 unicodedata.normalize("NFC", field_instance.value),
311 replacement_text="-",
312 )
313 elif (
314 field_instance.field.data_type == CustomField.FieldDataType.SELECT
315 and field_instance.field.extra_data["select_options"] is not None
316 ):
317 options = field_instance.field.extra_data["select_options"]
318 value = pathvalidate.sanitize_filename(
319 unicodedata.normalize(
320 "NFC",
321 next(
322 option["label"]
323 for option in options
324 if option["id"] == field_instance.value
325 ),
326 ),
327 replacement_text="-",
328 )
329 else:
330 value = field_instance.value
331 field_data["custom_fields"][
332 pathvalidate.sanitize_filename(
333 unicodedata.normalize("NFC", field_instance.field.name),
334 replacement_text="-",
335 )
336 ] = {
337 "type": type_,
338 "value": value,
339 }
340 return field_data
343def is_safe_relative_path(value: str) -> bool:
344 if value == "": 344 ↛ 345line 344 didn't jump to line 345 because the condition on line 344 was never true
345 return True
347 path = PurePath(value)
348 if path.is_absolute() or path.drive: 348 ↛ 349line 348 didn't jump to line 349 because the condition on line 348 was never true
349 return False
351 return ".." not in path.parts
354def validate_filepath_template_and_render(
355 template_string: str,
356 document: Document | None = None,
357) -> str | None:
358 """
359 Renders the given template string using either the given Document or using a dummy Document and data
361 Returns None if the string is not valid or an error occurred, otherwise
362 """
364 # Create the dummy document object with all fields filled in for validation purposes
365 if document is None: 365 ↛ 379line 365 didn't jump to line 379 because the condition on line 365 was always true
366 document = create_dummy_document()
367 tags_list = [Tag(name="Test Tag 1"), Tag(name="Another Test Tag")]
368 custom_fields = [
369 CustomFieldInstance(
370 field=CustomField(
371 name="Text Custom Field",
372 data_type=CustomField.FieldDataType.STRING,
373 ),
374 value_text="Some String Text",
375 ),
376 ]
377 else:
378 # or use the real document information
379 tags_list = document.tags.order_by("name").all()
380 custom_fields = CustomFieldInstance.global_objects.filter(document=document)
382 # Build the context dictionary
383 context = (
384 {"document": get_safe_document_context(document, tags=tags_list)}
385 | get_basic_metadata_context(document, no_value_default=NO_VALUE_PLACEHOLDER)
386 | get_creation_date_context(document)
387 | get_added_date_context(document)
388 | get_tags_context(tags_list)
389 | get_custom_fields_context(custom_fields)
390 )
392 # Try rendering the template
393 try:
394 # We load the custom tag used to remove spaces and newlines from the final string around the user string
395 template = _template_environment.from_string(
396 template_string,
397 template_class=FilePathTemplate,
398 )
399 rendered_template = template.render(context)
401 if not is_safe_relative_path(rendered_template): 401 ↛ 402line 401 didn't jump to line 402 because the condition on line 401 was never true
402 logger.warning(
403 "Template rendered an unsafe path (absolute or containing traversal).",
404 )
405 return None
407 # We're good!
408 return rendered_template
409 except UndefinedError:
410 # The undefined class logs this already for us
411 pass
412 except TemplateSyntaxError as e:
413 logger.warning(f"Template syntax error in filename generation: {e}")
414 except SecurityError as e:
415 logger.warning(f"Template attempted restricted operation: {e}")
416 except Exception as e:
417 logger.warning(f"Unknown error in filename generation: {e}")
418 logger.warning(
419 f"Invalid filename_format '{template_string}', falling back to default",
420 )
421 return None