Coverage for documents/templating/filepath.py: 73%

121 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1import logging 

2import os 

3import re 

4import unicodedata 

5from collections.abc import Iterable 

6from pathlib import PurePath 

7 

8import pathvalidate 

9from django.utils import timezone 

10from django.utils.text import slugify as django_slugify 

11from jinja2 import StrictUndefined 

12from jinja2 import Template 

13from jinja2 import TemplateSyntaxError 

14from jinja2 import UndefinedError 

15from jinja2 import make_logging_undefined 

16from jinja2.sandbox import SecurityError 

17 

18from documents.models import Correspondent 

19from documents.models import CustomField 

20from documents.models import CustomFieldInstance 

21from documents.models import Document 

22from documents.models import DocumentType 

23from documents.models import StoragePath 

24from documents.models import Tag 

25from documents.templating.environment import _template_environment 

26from documents.templating.filters import format_datetime 

27from documents.templating.filters import get_cf_value 

28from documents.templating.filters import localize_date 

29 

30logger = logging.getLogger("paperless.templating") 

31 

32_LogStrictUndefined = make_logging_undefined(logger, StrictUndefined) 

33 

34 

35class FilePathTemplate(Template): 

36 def render(self, *args, **kwargs) -> str: 

37 def clean_filepath(value: str) -> str: 

38 """ 

39 Clean up a filepath by: 

40 1. Normalizing Unicode to NFC form to prevent byte-level mismatches 

41 2. Removing newlines and carriage returns 

42 3. Removing extra spaces before and after forward slashes 

43 4. Preserving spaces in other parts of the path 

44 """ 

45 value = unicodedata.normalize("NFC", value) 

46 value = value.replace("\n", "").replace("\r", "") 

47 value = re.sub(r"\s*/\s*", "/", value) 

48 

49 # We remove trailing and leading separators, as these are always relative paths, not absolute, even if the user 

50 # tries 

51 return value.strip().strip(os.sep) 

52 

53 original_render = super().render(*args, **kwargs) 

54 

55 return clean_filepath(original_render) 

56 

57 

58class PlaceholderString(str): 

59 """ 

60 String subclass used as a sentinel for empty metadata values inside templates. 

61 

62 - Renders as \"-none-\" to preserve existing filename cleaning logic. 

63 - Compares equal to either \"-none-\" or \"none\" so templates can check for either. 

64 - Evaluates to False so {% if correspondent %} behaves intuitively. 

65 """ 

66 

67 def __new__(cls, value: str = "-none-"): 

68 return super().__new__(cls, value) 

69 

70 def __bool__(self) -> bool: 

71 return False 

72 

73 def __eq__(self, other) -> bool: 

74 if isinstance(other, str) and other == "none": 

75 other = "-none-" 

76 return super().__eq__(other) 

77 

78 def __ne__(self, other) -> bool: 

79 return not self.__eq__(other) 

80 

81 

82NO_VALUE_PLACEHOLDER = PlaceholderString("-none-") 

83 

84 

85class MatchingModelContext: 

86 """ 

87 Safe template context for related objects. 

88 

89 Keeps legacy behavior where including the object ina template yields the related object's 

90 name as a string, while still exposing limited attributes. 

91 """ 

92 

93 def __init__(self, *, id: int, name: str, path: str | None = None): 

94 self.id = id 

95 self.name = name 

96 self.path = path 

97 

98 def __str__(self) -> str: 

99 return self.name 

100 

101 

102_template_environment.undefined = _LogStrictUndefined 

103 

104_template_environment.filters["get_cf_value"] = get_cf_value 

105 

106_template_environment.filters["datetime"] = format_datetime 

107 

108_template_environment.filters["slugify"] = django_slugify 

109 

110_template_environment.filters["localize_date"] = localize_date 

111 

112 

113def create_dummy_document(): 

114 """ 

115 Create a dummy Document instance with all possible fields filled 

116 """ 

117 # Populate the document with representative values for every field 

118 dummy_doc = Document( 

119 pk=1, 

120 title="Sample Title", 

121 correspondent=Correspondent(name="Sample Correspondent"), 

122 storage_path=StoragePath(path="/dummy/path"), 

123 document_type=DocumentType(name="Sample Type"), 

124 content="This is some sample document content.", 

125 mime_type="application/pdf", 

126 checksum="dummychecksum12345678901234567890123456789012", 

127 archive_checksum="dummyarchivechecksum123456789012345678901234", 

128 page_count=5, 

129 created=timezone.now(), 

130 modified=timezone.now(), 

131 added=timezone.now(), 

132 filename="/dummy/filename.pdf", 

133 archive_filename="/dummy/archive_filename.pdf", 

134 original_filename="original_file.pdf", 

135 archive_serial_number=12345, 

136 ) 

137 return dummy_doc 

138 

139 

140def get_creation_date_context(document: Document) -> dict[str, str]: 

141 """ 

142 Given a Document, localizes the creation date and builds a context dictionary with some common, shorthand 

143 formatted values from it 

144 """ 

145 return { 

146 "created": document.created.isoformat(), 

147 "created_year": document.created.strftime("%Y"), 

148 "created_year_short": document.created.strftime("%y"), 

149 "created_month": document.created.strftime("%m"), 

150 "created_month_name": document.created.strftime("%B"), 

151 "created_month_name_short": document.created.strftime("%b"), 

152 "created_day": document.created.strftime("%d"), 

153 } 

154 

155 

156def get_added_date_context(document: Document) -> dict[str, str]: 

157 """ 

158 Given a Document, localizes the added date and builds a context dictionary with some common, shorthand 

159 formatted values from it 

160 """ 

161 local_added = timezone.localdate(document.added) 

162 

163 return { 

164 "added": local_added.isoformat(), 

165 "added_year": local_added.strftime("%Y"), 

166 "added_year_short": local_added.strftime("%y"), 

167 "added_month": local_added.strftime("%m"), 

168 "added_month_name": local_added.strftime("%B"), 

169 "added_month_name_short": local_added.strftime("%b"), 

170 "added_day": local_added.strftime("%d"), 

171 } 

172 

173 

174def get_basic_metadata_context( 

175 document: Document, 

176 *, 

177 no_value_default: str = NO_VALUE_PLACEHOLDER, 

178) -> dict[str, str]: 

179 """ 

180 Given a Document, constructs some basic information about it. If certain values are not set, 

181 they will be replaced with the no_value_default. 

182 

183 Regardless of set or not, the values will be sanitized 

184 """ 

185 return { 

186 "title": pathvalidate.sanitize_filename( 

187 unicodedata.normalize("NFC", document.title), 

188 replacement_text="-", 

189 ), 

190 "correspondent": pathvalidate.sanitize_filename( 

191 unicodedata.normalize("NFC", document.correspondent.name), 

192 replacement_text="-", 

193 ) 

194 if document.correspondent 

195 else no_value_default, 

196 "document_type": pathvalidate.sanitize_filename( 

197 unicodedata.normalize("NFC", document.document_type.name), 

198 replacement_text="-", 

199 ) 

200 if document.document_type 

201 else no_value_default, 

202 "asn": str(document.archive_serial_number) 

203 if document.archive_serial_number 

204 else no_value_default, 

205 "owner_username": document.owner.username 

206 if document.owner 

207 else no_value_default, 

208 "original_name": unicodedata.normalize( 

209 "NFC", 

210 PurePath(document.original_filename).with_suffix("").name, 

211 ) 

212 if document.original_filename 

213 else no_value_default, 

214 "doc_pk": f"{document.pk:07}", 

215 } 

216 

217 

218def get_safe_document_context( 

219 document: Document, 

220 tags: Iterable[Tag], 

221) -> dict[str, object]: 

222 """ 

223 Build a document context object to avoid supplying entire model instance. 

224 """ 

225 return { 

226 "id": document.pk, 

227 "pk": document.pk, 

228 "title": document.title, 

229 "content": document.content, 

230 "page_count": document.page_count, 

231 "created": document.created, 

232 "added": document.added, 

233 "modified": document.modified, 

234 "archive_serial_number": document.archive_serial_number, 

235 "mime_type": document.mime_type, 

236 "checksum": document.checksum, 

237 "archive_checksum": document.archive_checksum, 

238 "filename": document.filename, 

239 "archive_filename": document.archive_filename, 

240 "original_filename": document.original_filename, 

241 "owner": {"username": document.owner.username, "id": document.owner.id} 

242 if document.owner 

243 else None, 

244 "tags": [{"name": tag.name, "id": tag.id} for tag in tags], 

245 "correspondent": ( 

246 MatchingModelContext( 

247 name=document.correspondent.name, 

248 id=document.correspondent.id, 

249 ) 

250 if document.correspondent 

251 else None 

252 ), 

253 "document_type": ( 

254 MatchingModelContext( 

255 name=document.document_type.name, 

256 id=document.document_type.id, 

257 ) 

258 if document.document_type 

259 else None 

260 ), 

261 "storage_path": MatchingModelContext( 

262 name=document.storage_path.name, 

263 path=document.storage_path.path, 

264 id=document.storage_path.id, 

265 ) 

266 if document.storage_path 

267 else None, 

268 } 

269 

270 

271def get_tags_context(tags: Iterable[Tag]) -> dict[str, str | list[str]]: 

272 """ 

273 Given an Iterable of tags, constructs some context from them for usage 

274 """ 

275 return { 

276 "tag_list": pathvalidate.sanitize_filename( 

277 ",".join( 

278 sorted(unicodedata.normalize("NFC", tag.name) for tag in tags), 

279 ), 

280 replacement_text="-", 

281 ), 

282 # Assumed to be ordered, but a template could loop through to find what they want 

283 "tag_name_list": [unicodedata.normalize("NFC", x.name) for x in tags], 

284 } 

285 

286 

287def get_custom_fields_context( 

288 custom_fields: Iterable[CustomFieldInstance], 

289) -> dict[str, dict[str, dict[str, str]]]: 

290 """ 

291 Given an Iterable of CustomFieldInstance, builds a dictionary mapping the field name 

292 to its type and value 

293 """ 

294 field_data = {"custom_fields": {}} 

295 for field_instance in custom_fields: 

296 type_ = pathvalidate.sanitize_filename( 

297 field_instance.field.data_type, 

298 replacement_text="-", 

299 ) 

300 if field_instance.value is None: 300 ↛ 301line 300 didn't jump to line 301 because the condition on line 300 was never true

301 value = None 

302 # String types need to be sanitized 

303 elif field_instance.field.data_type in { 303 ↛ 313line 303 didn't jump to line 313 because the condition on line 303 was always true

304 CustomField.FieldDataType.MONETARY, 

305 CustomField.FieldDataType.STRING, 

306 CustomField.FieldDataType.URL, 

307 CustomField.FieldDataType.LONG_TEXT, 

308 }: 

309 value = pathvalidate.sanitize_filename( 

310 unicodedata.normalize("NFC", field_instance.value), 

311 replacement_text="-", 

312 ) 

313 elif ( 

314 field_instance.field.data_type == CustomField.FieldDataType.SELECT 

315 and field_instance.field.extra_data["select_options"] is not None 

316 ): 

317 options = field_instance.field.extra_data["select_options"] 

318 value = pathvalidate.sanitize_filename( 

319 unicodedata.normalize( 

320 "NFC", 

321 next( 

322 option["label"] 

323 for option in options 

324 if option["id"] == field_instance.value 

325 ), 

326 ), 

327 replacement_text="-", 

328 ) 

329 else: 

330 value = field_instance.value 

331 field_data["custom_fields"][ 

332 pathvalidate.sanitize_filename( 

333 unicodedata.normalize("NFC", field_instance.field.name), 

334 replacement_text="-", 

335 ) 

336 ] = { 

337 "type": type_, 

338 "value": value, 

339 } 

340 return field_data 

341 

342 

343def is_safe_relative_path(value: str) -> bool: 

344 if value == "": 344 ↛ 345line 344 didn't jump to line 345 because the condition on line 344 was never true

345 return True 

346 

347 path = PurePath(value) 

348 if path.is_absolute() or path.drive: 348 ↛ 349line 348 didn't jump to line 349 because the condition on line 348 was never true

349 return False 

350 

351 return ".." not in path.parts 

352 

353 

354def validate_filepath_template_and_render( 

355 template_string: str, 

356 document: Document | None = None, 

357) -> str | None: 

358 """ 

359 Renders the given template string using either the given Document or using a dummy Document and data 

360 

361 Returns None if the string is not valid or an error occurred, otherwise 

362 """ 

363 

364 # Create the dummy document object with all fields filled in for validation purposes 

365 if document is None: 365 ↛ 379line 365 didn't jump to line 379 because the condition on line 365 was always true

366 document = create_dummy_document() 

367 tags_list = [Tag(name="Test Tag 1"), Tag(name="Another Test Tag")] 

368 custom_fields = [ 

369 CustomFieldInstance( 

370 field=CustomField( 

371 name="Text Custom Field", 

372 data_type=CustomField.FieldDataType.STRING, 

373 ), 

374 value_text="Some String Text", 

375 ), 

376 ] 

377 else: 

378 # or use the real document information 

379 tags_list = document.tags.order_by("name").all() 

380 custom_fields = CustomFieldInstance.global_objects.filter(document=document) 

381 

382 # Build the context dictionary 

383 context = ( 

384 {"document": get_safe_document_context(document, tags=tags_list)} 

385 | get_basic_metadata_context(document, no_value_default=NO_VALUE_PLACEHOLDER) 

386 | get_creation_date_context(document) 

387 | get_added_date_context(document) 

388 | get_tags_context(tags_list) 

389 | get_custom_fields_context(custom_fields) 

390 ) 

391 

392 # Try rendering the template 

393 try: 

394 # We load the custom tag used to remove spaces and newlines from the final string around the user string 

395 template = _template_environment.from_string( 

396 template_string, 

397 template_class=FilePathTemplate, 

398 ) 

399 rendered_template = template.render(context) 

400 

401 if not is_safe_relative_path(rendered_template): 401 ↛ 402line 401 didn't jump to line 402 because the condition on line 401 was never true

402 logger.warning( 

403 "Template rendered an unsafe path (absolute or containing traversal).", 

404 ) 

405 return None 

406 

407 # We're good! 

408 return rendered_template 

409 except UndefinedError: 

410 # The undefined class logs this already for us 

411 pass 

412 except TemplateSyntaxError as e: 

413 logger.warning(f"Template syntax error in filename generation: {e}") 

414 except SecurityError as e: 

415 logger.warning(f"Template attempted restricted operation: {e}") 

416 except Exception as e: 

417 logger.warning(f"Unknown error in filename generation: {e}") 

418 logger.warning( 

419 f"Invalid filename_format '{template_string}', falling back to default", 

420 ) 

421 return None