Coverage for paperless/settings/__init__.py: 87%
375 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1import datetime
2import json
3import logging
4import logging.config
5import math
6import multiprocessing
7import os
8import tempfile
9from pathlib import Path
10from typing import Any
11from typing import Final
12from urllib.parse import urlparse
14from django.core.exceptions import ImproperlyConfigured
15from django.utils.translation import gettext_lazy as _
16from dotenv import load_dotenv
18from paperless.settings.custom import parse_beat_schedule
19from paperless.settings.custom import parse_dateparser_languages
20from paperless.settings.custom import parse_db_settings
21from paperless.settings.custom import parse_hosting_settings
22from paperless.settings.custom import parse_ignore_dates
23from paperless.settings.custom import parse_redis_url
24from paperless.settings.parsers import get_bool_from_env
25from paperless.settings.parsers import get_choice_from_env
26from paperless.settings.parsers import get_float_from_env
27from paperless.settings.parsers import get_int_from_env
28from paperless.settings.parsers import get_list_from_env
29from paperless.settings.parsers import get_path_from_env
31logger = logging.getLogger("paperless.settings")
33# Tap paperless.conf if it's available
34for path in [
35 os.getenv("PAPERLESS_CONFIGURATION_PATH"),
36 "../paperless.conf",
37 "/etc/paperless.conf",
38 "/usr/local/etc/paperless.conf",
39]:
40 if path and Path(path).exists(): 40 ↛ 41line 40 didn't jump to line 41 because the condition on line 40 was never true
41 load_dotenv(path)
42 break
44# There are multiple levels of concurrency in paperless:
45# - Multiple consumers may be run in parallel.
46# - Each consumer may process multiple pages in parallel.
47# - Each Tesseract OCR run may spawn multiple threads to process a single page
48# slightly faster.
49# The performance gains from having tesseract use multiple threads are minimal.
50# However, when multiple pages are processed in parallel, the total number of
51# OCR threads may exceed the number of available cpu cores, which will
52# dramatically slow down the consumption process. This settings limits each
53# Tesseract process to one thread.
54os.environ["OMP_THREAD_LIMIT"] = "1"
57# NEVER RUN WITH DEBUG IN PRODUCTION.
58DEBUG = get_bool_from_env("PAPERLESS_DEBUG", "NO")
61###############################################################################
62# Directories #
63###############################################################################
65BASE_DIR: Path = Path(__file__).resolve().parent.parent.parent
67STATIC_ROOT = get_path_from_env("PAPERLESS_STATICDIR", BASE_DIR.parent / "static")
69MEDIA_ROOT = get_path_from_env("PAPERLESS_MEDIA_ROOT", BASE_DIR.parent / "media")
70ORIGINALS_DIR = MEDIA_ROOT / "documents" / "originals"
71ARCHIVE_DIR = MEDIA_ROOT / "documents" / "archive"
72THUMBNAIL_DIR = MEDIA_ROOT / "documents" / "thumbnails"
73SHARE_LINK_BUNDLE_DIR = MEDIA_ROOT / "documents" / "share_link_bundles"
75DATA_DIR = get_path_from_env("PAPERLESS_DATA_DIR", BASE_DIR.parent / "data")
77# Check deprecated setting first
78EMPTY_TRASH_DIR = (
79 get_path_from_env("PAPERLESS_TRASH_DIR", os.getenv("PAPERLESS_EMPTY_TRASH_DIR"))
80 if os.getenv("PAPERLESS_TRASH_DIR") or os.getenv("PAPERLESS_EMPTY_TRASH_DIR")
81 else None
82)
84# Lock file for synchronizing changes to the MEDIA directory across multiple
85# threads.
86MEDIA_LOCK = MEDIA_ROOT / "media.lock"
87INDEX_DIR = DATA_DIR / "index"
89ADVANCED_FUZZY_SEARCH_THRESHOLD: float | None = get_float_from_env(
90 "PAPERLESS_ADVANCED_FUZZY_SEARCH_THRESHOLD",
91)
93MODEL_FILE = get_path_from_env(
94 "PAPERLESS_MODEL_FILE",
95 DATA_DIR / "classification_model.pickle",
96)
98# Minimum confidence (0.0-1.0) for the ML classifier to assign a correspondent,
99# document type, or storage path. 0.0 disables the threshold.
100CLASSIFIER_MATCH_THRESHOLD: Final[float] = get_float_from_env(
101 "PAPERLESS_CLASSIFIER_MATCH_THRESHOLD",
102 0.3,
103)
104MATCH_REGEX_TIMEOUT_SECONDS: Final[float] = get_float_from_env(
105 "PAPERLESS_MATCH_REGEX_TIMEOUT_SECONDS",
106 0.1,
107)
108LLM_INDEX_DIR = DATA_DIR / "llm_index"
109LLM_INDEX_LOCK = LLM_INDEX_DIR / "index.lock"
110# Cross-process read/write lock guarding the LLM index compaction/migration
111# file swap. Readers hold it shared; the swap takes it exclusively so it never
112# runs while a reader connection is open. Must be a SQLite (.db) file.
113LLM_INDEX_RWLOCK = LLM_INDEX_DIR / "llmindex.rwlock.db"
114# Seconds the compaction swap waits for active readers to drain before skipping
115# this cycle (it is a maintenance operation; the next run retries).
116LLM_INDEX_COMPACTION_LOCK_TIMEOUT = 30
118LOGGING_DIR = get_path_from_env("PAPERLESS_LOGGING_DIR", DATA_DIR / "log")
120CONSUMPTION_DIR = get_path_from_env(
121 "PAPERLESS_CONSUMPTION_DIR",
122 BASE_DIR.parent / "consume",
123)
125# This will be created if it doesn't exist
126SCRATCH_DIR = get_path_from_env(
127 "PAPERLESS_SCRATCH_DIR",
128 Path(tempfile.gettempdir()) / "paperless",
129)
131###############################################################################
132# Application Definition #
133###############################################################################
135env_apps = get_list_from_env("PAPERLESS_APPS")
137INSTALLED_APPS = [
138 "whitenoise.runserver_nostatic",
139 "django.contrib.auth",
140 "django.contrib.contenttypes",
141 "django.contrib.sessions",
142 "django.contrib.messages",
143 "django.contrib.staticfiles",
144 "corsheaders",
145 "django_extensions",
146 "paperless",
147 "documents.apps.DocumentsConfig",
148 "paperless_mail.apps.PaperlessMailConfig",
149 "django.contrib.admin",
150 "rest_framework",
151 "rest_framework.authtoken",
152 "django_filters",
153 "guardian",
154 "allauth",
155 "allauth.account",
156 "allauth.socialaccount",
157 "allauth.mfa",
158 "allauth.headless",
159 "drf_spectacular",
160 "drf_spectacular_sidecar",
161 "treenode",
162 *env_apps,
163]
165if DEBUG: 165 ↛ 166line 165 didn't jump to line 166 because the condition on line 165 was never true
166 INSTALLED_APPS.append("channels")
168REST_FRAMEWORK = {
169 "DEFAULT_AUTHENTICATION_CLASSES": [
170 "paperless.auth.PaperlessBasicAuthentication",
171 "rest_framework.authentication.TokenAuthentication",
172 "rest_framework.authentication.SessionAuthentication",
173 ],
174 "DEFAULT_VERSIONING_CLASS": "rest_framework.versioning.AcceptHeaderVersioning",
175 "DEFAULT_VERSION": "10", # match src-ui/src/environments/environment.prod.ts
176 # Make sure these are ordered and that the most recent version appears
177 # last. See api.md#api-versioning when adding new versions.
178 "ALLOWED_VERSIONS": ["9", "10"],
179 # DRF Spectacular default schema
180 "DEFAULT_SCHEMA_CLASS": "drf_spectacular.openapi.AutoSchema",
181 "DEFAULT_THROTTLE_RATES": {
182 "login": os.getenv("PAPERLESS_TOKEN_THROTTLE_RATE", "5/min"),
183 },
184}
186if DEBUG: 186 ↛ 187line 186 didn't jump to line 187 because the condition on line 186 was never true
187 REST_FRAMEWORK["DEFAULT_AUTHENTICATION_CLASSES"].append(
188 "paperless.auth.AngularApiAuthenticationOverride",
189 )
191MIDDLEWARE = [
192 "django.middleware.security.SecurityMiddleware",
193 "whitenoise.middleware.WhiteNoiseMiddleware",
194 "django.contrib.sessions.middleware.SessionMiddleware",
195 "corsheaders.middleware.CorsMiddleware",
196 "django.middleware.locale.LocaleMiddleware",
197 "django.middleware.common.CommonMiddleware",
198 "django.middleware.csrf.CsrfViewMiddleware",
199 "paperless.middleware.ApiVersionMiddleware",
200 "django.contrib.auth.middleware.AuthenticationMiddleware",
201 "django.contrib.messages.middleware.MessageMiddleware",
202 "django.middleware.clickjacking.XFrameOptionsMiddleware",
203 "allauth.account.middleware.AccountMiddleware",
204]
206# Optional to enable compression. The subclass leaves server-sent events
207# uncompressed; see paperless.middleware.StreamAwareCompressionMiddleware.
208if get_bool_from_env("PAPERLESS_ENABLE_COMPRESSION", "yes"): # pragma: no cover 208 ↛ 211line 208 didn't jump to line 211 because the condition on line 208 was always true
209 MIDDLEWARE.insert(0, "paperless.middleware.StreamAwareCompressionMiddleware")
211ROOT_URLCONF = "paperless.urls"
214FORCE_SCRIPT_NAME, BASE_URL, LOGIN_URL, LOGIN_REDIRECT_URL, LOGOUT_REDIRECT_URL = (
215 parse_hosting_settings()
216)
218# DRF Spectacular settings
219SPECTACULAR_SETTINGS = {
220 "TITLE": "Paperless-ngx REST API",
221 "DESCRIPTION": "OpenAPI Spec for Paperless-ngx",
222 "VERSION": "6.0.0",
223 "SERVE_INCLUDE_SCHEMA": False,
224 "SWAGGER_UI_DIST": "SIDECAR",
225 "COMPONENT_SPLIT_REQUEST": True,
226 "EXTERNAL_DOCS": {
227 "description": "Paperless-ngx API Documentation",
228 "url": "https://docs.paperless-ngx.com/api/",
229 },
230 "ENUM_NAME_OVERRIDES": {
231 "MatchingAlgorithm": "documents.models.MatchingModel.MATCHING_ALGORITHMS",
232 "BarcodeFormatEnum": "documents.models.DocumentBarcode.Format",
233 },
234 "SCHEMA_PATH_PREFIX_INSERT": FORCE_SCRIPT_NAME or "",
235}
237WSGI_APPLICATION = "paperless.wsgi.application"
238ASGI_APPLICATION = "paperless.asgi.application"
240STATIC_URL = os.getenv("PAPERLESS_STATIC_URL", BASE_URL + "static/")
241WHITENOISE_STATIC_PREFIX = "/static/"
243STORAGES = {
244 "staticfiles": {
245 "BACKEND": "whitenoise.storage.CompressedStaticFilesStorage",
246 },
247 "default": {"BACKEND": "django.core.files.storage.FileSystemStorage"},
248}
250_CELERY_REDIS_URL, _CHANNELS_REDIS_URL = parse_redis_url(
251 os.getenv("PAPERLESS_REDIS", None),
252)
253_REDIS_KEY_PREFIX = os.getenv("PAPERLESS_REDIS_PREFIX", "")
255TEMPLATES = [
256 {
257 "BACKEND": "django.template.backends.django.DjangoTemplates",
258 "DIRS": [],
259 "APP_DIRS": True,
260 "OPTIONS": {
261 "context_processors": [
262 "django.template.context_processors.debug",
263 "django.template.context_processors.request",
264 "django.contrib.auth.context_processors.auth",
265 "django.contrib.messages.context_processors.messages",
266 "documents.context_processors.settings",
267 ],
268 },
269 },
270]
272_CHANNELS_BACKEND = os.environ.get(
273 "PAPERLESS_CHANNELS_BACKEND",
274 "channels_redis.pubsub.RedisPubSubChannelLayer",
275)
276CHANNEL_LAYERS = {
277 "default": {
278 "BACKEND": _CHANNELS_BACKEND,
279 },
280}
282if _CHANNELS_BACKEND.startswith("channels_redis."): 282 ↛ 294line 282 didn't jump to line 294 because the condition on line 282 was always true
283 CHANNEL_LAYERS["default"]["CONFIG"] = {
284 "hosts": [_CHANNELS_REDIS_URL],
285 "capacity": 2000, # default 100
286 "expiry": 15, # default 60
287 "prefix": _REDIS_KEY_PREFIX,
288 }
290###############################################################################
291# Email (SMTP) Backend #
292###############################################################################
294EMAIL_HOST: Final[str] = os.getenv("PAPERLESS_EMAIL_HOST", "localhost")
295EMAIL_PORT: Final[int] = get_int_from_env("PAPERLESS_EMAIL_PORT", 25)
296EMAIL_HOST_USER: Final[str] = os.getenv("PAPERLESS_EMAIL_HOST_USER", "")
297EMAIL_HOST_PASSWORD: Final[str] = os.getenv("PAPERLESS_EMAIL_HOST_PASSWORD", "")
298DEFAULT_FROM_EMAIL: Final[str] = os.getenv("PAPERLESS_EMAIL_FROM", EMAIL_HOST_USER)
299EMAIL_USE_TLS: Final[bool] = get_bool_from_env("PAPERLESS_EMAIL_USE_TLS")
300EMAIL_USE_SSL: Final[bool] = get_bool_from_env("PAPERLESS_EMAIL_USE_SSL")
301EMAIL_SUBJECT_PREFIX: Final[str] = "[Paperless-ngx] "
302EMAIL_TIMEOUT = 30.0
303EMAIL_ENABLED = EMAIL_HOST != "localhost" or EMAIL_HOST_USER != ""
304if DEBUG: # pragma: no cover 304 ↛ 305line 304 didn't jump to line 305 because the condition on line 304 was never true
305 EMAIL_BACKEND = "django.core.mail.backends.filebased.EmailBackend"
306 EMAIL_FILE_PATH = BASE_DIR / "sent_emails"
308###############################################################################
309# Security #
310###############################################################################
312AUTHENTICATION_BACKENDS = [
313 "guardian.backends.ObjectPermissionBackend",
314 "django.contrib.auth.backends.ModelBackend",
315 "allauth.account.auth_backends.AuthenticationBackend",
316]
318ACCOUNT_LOGOUT_ON_GET = True
319ACCOUNT_DEFAULT_HTTP_PROTOCOL = os.getenv(
320 "PAPERLESS_ACCOUNT_DEFAULT_HTTP_PROTOCOL",
321 "https",
322)
324ACCOUNT_ADAPTER = "paperless.adapter.CustomAccountAdapter"
325ACCOUNT_ALLOW_SIGNUPS = get_bool_from_env("PAPERLESS_ACCOUNT_ALLOW_SIGNUPS")
326ACCOUNT_DEFAULT_GROUPS = get_list_from_env("PAPERLESS_ACCOUNT_DEFAULT_GROUPS")
328SOCIALACCOUNT_ADAPTER = "paperless.adapter.CustomSocialAccountAdapter"
329SOCIALACCOUNT_ALLOW_SIGNUPS = get_bool_from_env(
330 "PAPERLESS_SOCIALACCOUNT_ALLOW_SIGNUPS",
331 "yes",
332)
333SOCIALACCOUNT_AUTO_SIGNUP = get_bool_from_env("PAPERLESS_SOCIAL_AUTO_SIGNUP")
334SOCIALACCOUNT_PROVIDERS = json.loads(
335 os.getenv("PAPERLESS_SOCIALACCOUNT_PROVIDERS", "{}"),
336)
337SOCIAL_ACCOUNT_DEFAULT_GROUPS = get_list_from_env(
338 "PAPERLESS_SOCIAL_ACCOUNT_DEFAULT_GROUPS",
339)
340SOCIAL_ACCOUNT_SYNC_GROUPS = get_bool_from_env("PAPERLESS_SOCIAL_ACCOUNT_SYNC_GROUPS")
341SOCIAL_ACCOUNT_SYNC_GROUPS_CLAIM: Final[str] = os.getenv(
342 "PAPERLESS_SOCIAL_ACCOUNT_SYNC_GROUPS_CLAIM",
343 "groups",
344)
345SOCIAL_ACCOUNT_SYNC_SUPERUSER_GROUP: Final[str | None] = os.getenv(
346 "PAPERLESS_SOCIAL_ACCOUNT_SYNC_SUPERUSER_GROUP",
347)
348SOCIAL_ACCOUNT_SYNC_STAFF_GROUP: Final[str | None] = os.getenv(
349 "PAPERLESS_SOCIAL_ACCOUNT_SYNC_STAFF_GROUP",
350)
352HEADLESS_TOKEN_STRATEGY = "paperless.adapter.DrfTokenStrategy"
354MFA_TOTP_ISSUER = "Paperless-ngx"
356ACCOUNT_EMAIL_SUBJECT_PREFIX = "[Paperless-ngx] "
358DISABLE_REGULAR_LOGIN = get_bool_from_env("PAPERLESS_DISABLE_REGULAR_LOGIN")
359REDIRECT_LOGIN_TO_SSO = get_bool_from_env("PAPERLESS_REDIRECT_LOGIN_TO_SSO")
361AUTO_LOGIN_USERNAME = os.getenv("PAPERLESS_AUTO_LOGIN_USERNAME")
363ACCOUNT_EMAIL_VERIFICATION = (
364 "none"
365 if not EMAIL_ENABLED
366 else os.getenv(
367 "PAPERLESS_ACCOUNT_EMAIL_VERIFICATION",
368 "optional",
369 )
370)
372ACCOUNT_EMAIL_UNKNOWN_ACCOUNTS = get_bool_from_env(
373 "PAPERLESS_ACCOUNT_EMAIL_UNKNOWN_ACCOUNTS",
374 "True",
375)
377ACCOUNT_SESSION_REMEMBER = get_bool_from_env(
378 "PAPERLESS_ACCOUNT_SESSION_REMEMBER",
379 "True",
380)
381SESSION_EXPIRE_AT_BROWSER_CLOSE = not ACCOUNT_SESSION_REMEMBER
382SESSION_COOKIE_AGE = get_int_from_env(
383 "PAPERLESS_SESSION_COOKIE_AGE",
384 60 * 60 * 24 * 7 * 3,
385)
386# https://docs.djangoproject.com/en/5.1/ref/settings/#std-setting-SESSION_ENGINE
387SESSION_ENGINE = "django.contrib.sessions.backends.cached_db"
389if AUTO_LOGIN_USERNAME: 389 ↛ 390line 389 didn't jump to line 390 because the condition on line 389 was never true
390 _index = MIDDLEWARE.index("django.contrib.auth.middleware.AuthenticationMiddleware")
391 # This overrides everything the auth middleware is doing but still allows
392 # regular login in case the provided user does not exist.
393 MIDDLEWARE.insert(_index + 1, "paperless.auth.AutoLoginMiddleware")
396def _parse_remote_user_settings() -> str:
397 global MIDDLEWARE, AUTHENTICATION_BACKENDS, REST_FRAMEWORK
398 enable = get_bool_from_env("PAPERLESS_ENABLE_HTTP_REMOTE_USER")
399 enable_api = get_bool_from_env("PAPERLESS_ENABLE_HTTP_REMOTE_USER_API")
400 if enable or enable_api: 400 ↛ 401line 400 didn't jump to line 401 because the condition on line 400 was never true
401 MIDDLEWARE.append("paperless.auth.HttpRemoteUserMiddleware")
402 AUTHENTICATION_BACKENDS.insert(
403 0,
404 "django.contrib.auth.backends.RemoteUserBackend",
405 )
407 if enable_api: 407 ↛ 408line 407 didn't jump to line 408 because the condition on line 407 was never true
408 REST_FRAMEWORK["DEFAULT_AUTHENTICATION_CLASSES"].insert(
409 0,
410 "paperless.auth.PaperlessRemoteUserAuthentication",
411 )
413 header_name = os.getenv(
414 "PAPERLESS_HTTP_REMOTE_USER_HEADER_NAME",
415 "HTTP_REMOTE_USER",
416 )
418 return header_name
421HTTP_REMOTE_USER_HEADER_NAME = _parse_remote_user_settings()
423# X-Frame options for embedded PDF display:
424X_FRAME_OPTIONS = "SAMEORIGIN"
426# The next 3 settings can also be set using just PAPERLESS_URL
427CSRF_TRUSTED_ORIGINS = get_list_from_env("PAPERLESS_CSRF_TRUSTED_ORIGINS")
429if DEBUG: 429 ↛ 431line 429 didn't jump to line 431 because the condition on line 429 was never true
430 # Allow access from the angular development server during debugging
431 CSRF_TRUSTED_ORIGINS.append("http://localhost:4200")
433# We allow CORS from localhost:8000
434CORS_ALLOWED_ORIGINS = get_list_from_env(
435 "PAPERLESS_CORS_ALLOWED_HOSTS",
436 default=["http://localhost:8000"],
437)
439if DEBUG: 439 ↛ 441line 439 didn't jump to line 441 because the condition on line 439 was never true
440 # Allow access from the angular development server during debugging
441 CORS_ALLOWED_ORIGINS.append("http://localhost:4200")
443CORS_ALLOW_CREDENTIALS = True
445CORS_EXPOSE_HEADERS = [
446 "Content-Disposition",
447]
449ALLOWED_HOSTS = get_list_from_env("PAPERLESS_ALLOWED_HOSTS", default=["*"])
450if ALLOWED_HOSTS != ["*"]: 450 ↛ 452line 450 didn't jump to line 452 because the condition on line 450 was never true
451 # always allow localhost. Necessary e.g. for healthcheck in docker.
452 ALLOWED_HOSTS.append("localhost")
455def _parse_paperless_url():
456 global CSRF_TRUSTED_ORIGINS, CORS_ALLOWED_ORIGINS, ALLOWED_HOSTS
457 url = os.getenv("PAPERLESS_URL")
458 if url: 458 ↛ 463line 458 didn't jump to line 463 because the condition on line 458 was always true
459 CSRF_TRUSTED_ORIGINS.append(url)
460 CORS_ALLOWED_ORIGINS.append(url)
461 ALLOWED_HOSTS.append(urlparse(url).hostname)
463 return url
466PAPERLESS_URL = _parse_paperless_url()
469def _get_allauth_trusted_proxy_count(trusted_proxies: list[str]) -> int:
470 count = get_int_from_env(
471 "PAPERLESS_ALLAUTH_TRUSTED_PROXY_COUNT",
472 len(trusted_proxies),
473 )
474 if count < 0: 474 ↛ 475line 474 didn't jump to line 475 because the condition on line 474 was never true
475 raise ImproperlyConfigured(
476 "PAPERLESS_ALLAUTH_TRUSTED_PROXY_COUNT must be zero or greater",
477 )
478 return count
481# For use with trusted proxies
482TRUSTED_PROXIES = get_list_from_env("PAPERLESS_TRUSTED_PROXIES")
483ALLAUTH_TRUSTED_PROXY_COUNT = _get_allauth_trusted_proxy_count(TRUSTED_PROXIES)
484ALLAUTH_TRUSTED_CLIENT_IP_HEADER = os.getenv(
485 "PAPERLESS_ALLAUTH_TRUSTED_CLIENT_IP_HEADER",
486)
488USE_X_FORWARDED_HOST = get_bool_from_env("PAPERLESS_USE_X_FORWARD_HOST", "false")
489USE_X_FORWARDED_PORT = get_bool_from_env("PAPERLESS_USE_X_FORWARD_PORT", "false")
490SECURE_PROXY_SSL_HEADER = (
491 tuple(json.loads(os.environ["PAPERLESS_PROXY_SSL_HEADER"]))
492 if "PAPERLESS_PROXY_SSL_HEADER" in os.environ
493 else None
494)
496SECRET_KEY = os.getenv("PAPERLESS_SECRET_KEY")
497if not (SECRET_KEY or "").strip() or SECRET_KEY == "change-me": # pragma: no cover 497 ↛ 498line 497 didn't jump to line 498 because the condition on line 497 was never true
498 raise ImproperlyConfigured(
499 "PAPERLESS_SECRET_KEY is not set or is the default 'change-me' value. "
500 "A unique, secret key is required for secure operation. "
501 'Generate one with: python3 -c "import secrets; print(secrets.token_urlsafe(64))"',
502 )
504AUTH_PASSWORD_VALIDATORS = [
505 {
506 "NAME": "django.contrib.auth.password_validation.UserAttributeSimilarityValidator",
507 },
508 {
509 "NAME": "django.contrib.auth.password_validation.MinimumLengthValidator",
510 },
511 {
512 "NAME": "django.contrib.auth.password_validation.CommonPasswordValidator",
513 },
514 {
515 "NAME": "django.contrib.auth.password_validation.NumericPasswordValidator",
516 },
517]
519# Disable Django's artificial limit on the number of form fields to submit at
520# once. This is a protection against overloading the server, but since this is
521# a self-hosted sort of gig, the benefits of being able to mass-delete a ton
522# of log entries outweigh the benefits of such a safeguard.
524DATA_UPLOAD_MAX_NUMBER_FIELDS = None
526COOKIE_PREFIX = os.getenv("PAPERLESS_COOKIE_PREFIX", "")
528CSRF_COOKIE_NAME = f"{COOKIE_PREFIX}csrftoken"
529SESSION_COOKIE_NAME = f"{COOKIE_PREFIX}sessionid"
530LANGUAGE_COOKIE_NAME = f"{COOKIE_PREFIX}django_language"
532EMAIL_CERTIFICATE_FILE = get_path_from_env("PAPERLESS_EMAIL_CERTIFICATE_LOCATION")
533EMAIL_ALLOW_INTERNAL_HOSTS = get_bool_from_env(
534 "PAPERLESS_EMAIL_ALLOW_INTERNAL_HOSTS",
535 "true",
536)
539###############################################################################
540# Database #
541###############################################################################
543DATABASES = parse_db_settings(DATA_DIR)
545if os.getenv("PAPERLESS_DBENGINE") == "mariadb": 545 ↛ 550line 545 didn't jump to line 550 because the condition on line 545 was never true
546 # Silence Django error on old MariaDB versions.
547 # VARCHAR can support > 255 in modern versions
548 # https://docs.djangoproject.com/en/4.1/ref/checks/#database
549 # https://mariadb.com/kb/en/innodb-system-variables/#innodb_large_prefix
550 SILENCED_SYSTEM_CHECKS = ["mysql.W003"]
552DEFAULT_AUTO_FIELD = "django.db.models.AutoField"
554###############################################################################
555# Internationalization #
556###############################################################################
558LANGUAGE_CODE = "en-us"
560LANGUAGES = [
561 ("en-us", _("English (US)")), # needs to be first to act as fallback language
562 ("ar-ar", _("Arabic")),
563 ("af-za", _("Afrikaans")),
564 ("be-by", _("Belarusian")),
565 ("bg-bg", _("Bulgarian")),
566 ("ca-es", _("Catalan")),
567 ("cs-cz", _("Czech")),
568 ("da-dk", _("Danish")),
569 ("de-de", _("German")),
570 ("el-gr", _("Greek")),
571 ("en-gb", _("English (GB)")),
572 ("es-es", _("Spanish")),
573 ("fa-ir", _("Persian")),
574 ("fi-fi", _("Finnish")),
575 ("fr-fr", _("French")),
576 ("hu-hu", _("Hungarian")),
577 ("id-id", _("Indonesian")),
578 ("it-it", _("Italian")),
579 ("ja-jp", _("Japanese")),
580 ("ko-kr", _("Korean")),
581 ("lb-lu", _("Luxembourgish")),
582 ("no-no", _("Norwegian")),
583 ("nl-nl", _("Dutch")),
584 ("pl-pl", _("Polish")),
585 ("pt-br", _("Portuguese (Brazil)")),
586 ("pt-pt", _("Portuguese")),
587 ("ro-ro", _("Romanian")),
588 ("ru-ru", _("Russian")),
589 ("sk-sk", _("Slovak")),
590 ("sl-si", _("Slovenian")),
591 ("sr-cs", _("Serbian")),
592 ("sv-se", _("Swedish")),
593 ("tr-tr", _("Turkish")),
594 ("uk-ua", _("Ukrainian")),
595 ("vi-vn", _("Vietnamese")),
596 ("zh-cn", _("Chinese Simplified")),
597 ("zh-tw", _("Chinese Traditional")),
598]
600LOCALE_PATHS = [BASE_DIR / "locale"]
602TIME_ZONE = os.getenv("PAPERLESS_TIME_ZONE", "UTC")
604USE_I18N = True
606USE_L10N = True
608USE_TZ = True
610###############################################################################
611# Logging #
612###############################################################################
614LOGGING_DIR.mkdir(parents=True, exist_ok=True)
616LOGROTATE_MAX_SIZE = get_int_from_env("PAPERLESS_LOGROTATE_MAX_SIZE", 1024 * 1024)
617LOGROTATE_MAX_BACKUPS = get_int_from_env("PAPERLESS_LOGROTATE_MAX_BACKUPS", 20)
619LOGGING = {
620 "version": 1,
621 "disable_existing_loggers": False,
622 "formatters": {
623 "verbose": {
624 "()": "paperless.logging.ConsumeTaskFormatter",
625 },
626 "simple": {
627 "format": "{levelname} {message}",
628 "style": "{",
629 },
630 },
631 "handlers": {
632 "console": {
633 "level": "DEBUG" if DEBUG else "INFO",
634 "class": "logging.StreamHandler",
635 "formatter": "verbose",
636 },
637 "file_paperless": {
638 "class": "concurrent_log_handler.ConcurrentRotatingFileHandler",
639 "formatter": "verbose",
640 "filename": LOGGING_DIR / "paperless.log",
641 "maxBytes": LOGROTATE_MAX_SIZE,
642 "backupCount": LOGROTATE_MAX_BACKUPS,
643 },
644 "file_mail": {
645 "class": "concurrent_log_handler.ConcurrentRotatingFileHandler",
646 "formatter": "verbose",
647 "filename": LOGGING_DIR / "mail.log",
648 "maxBytes": LOGROTATE_MAX_SIZE,
649 "backupCount": LOGROTATE_MAX_BACKUPS,
650 },
651 "file_celery": {
652 "class": "concurrent_log_handler.ConcurrentRotatingFileHandler",
653 "formatter": "verbose",
654 "filename": LOGGING_DIR / "celery.log",
655 "maxBytes": LOGROTATE_MAX_SIZE,
656 "backupCount": LOGROTATE_MAX_BACKUPS,
657 },
658 },
659 "root": {"handlers": ["console"]},
660 "loggers": {
661 "paperless": {"handlers": ["file_paperless"], "level": "DEBUG"},
662 "paperless_mail": {"handlers": ["file_mail"], "level": "DEBUG"},
663 "paperless_ai": {"handlers": ["file_paperless"], "level": "DEBUG"},
664 "ocrmypdf": {"handlers": ["file_paperless"], "level": "INFO"},
665 "celery": {"handlers": ["file_celery"], "level": "DEBUG"},
666 "kombu": {"handlers": ["file_celery"], "level": "DEBUG"},
667 "_granian": {"handlers": ["file_paperless"], "level": "DEBUG"},
668 "granian.access": {"handlers": ["file_paperless"], "level": "DEBUG"},
669 "httpx": {"level": "WARNING"},
670 },
671}
673# Configure logging before calling any logger in settings.py so it will respect the log format, even if Django has not parsed the settings yet.
674logging.config.dictConfig(LOGGING)
677###############################################################################
678# Task queue #
679###############################################################################
681# https://docs.celeryq.dev/en/stable/userguide/configuration.html
683CELERY_BROKER_URL = _CELERY_REDIS_URL
684CELERY_RESULT_BACKEND = _CELERY_REDIS_URL
685CELERY_RESULT_SERIALIZER = "signed-pickle"
686# Results are only needed for chord synchronization
687# a short TTL avoids Redis memory accumulation.
688CELERY_RESULT_EXPIRES = 3600
689CELERY_TIMEZONE = TIME_ZONE
691CELERY_WORKER_HIJACK_ROOT_LOGGER = False
692CELERY_WORKER_CONCURRENCY: Final[int] = get_int_from_env("PAPERLESS_TASK_WORKERS", 1)
693TASK_WORKERS = CELERY_WORKER_CONCURRENCY
694CELERY_WORKER_MAX_TASKS_PER_CHILD = 1
695CELERY_WORKER_SEND_TASK_EVENTS = True
696CELERY_TASK_SEND_SENT_EVENT = True
697CELERY_SEND_TASK_SENT_EVENT = True
698CELERY_BROKER_CONNECTION_RETRY = True
699CELERY_BROKER_CONNECTION_RETRY_ON_STARTUP = True
700CELERY_BROKER_TRANSPORT_OPTIONS = {
701 "global_keyprefix": _REDIS_KEY_PREFIX,
702}
703CELERY_RESULT_BACKEND_TRANSPORT_OPTIONS = {
704 "global_keyprefix": _REDIS_KEY_PREFIX,
705}
707CELERY_TASK_TRACK_STARTED = True
708CELERY_TASK_TIME_LIMIT: Final[int] = get_int_from_env("PAPERLESS_WORKER_TIMEOUT", 1800)
710# https://docs.celeryq.dev/en/stable/userguide/configuration.html#std-setting-task_allow_error_cb_on_chord_header
711# Without this, a failing chord header never triggers the errback, so a mail
712# whose attachments all fail is never recorded and is re-fetched forever.
713# The errback runs once per failed header task, so it must be idempotent.
714CELERY_TASK_ALLOW_ERROR_CB_ON_CHORD_HEADER = True
716CELERY_CACHE_BACKEND = "default"
718# https://docs.celeryq.dev/en/stable/userguide/configuration.html#task-serializer
719# Uses HMAC-signed pickle to prevent RCE via malicious messages on an exposed Redis broker.
720# The signed-pickle serializer is registered in paperless/celery.py.
721CELERY_TASK_SERIALIZER = "signed-pickle"
722# https://docs.celeryq.dev/en/stable/userguide/configuration.html#std-setting-accept_content
723CELERY_ACCEPT_CONTENT = ["application/json", "application/x-signed-pickle"]
725# https://docs.celeryq.dev/en/stable/userguide/configuration.html#beat-schedule
726CELERY_BEAT_SCHEDULE = parse_beat_schedule()
728# https://docs.celeryq.dev/en/stable/userguide/configuration.html#beat-schedule-filename
729CELERY_BEAT_SCHEDULE_FILENAME = str(DATA_DIR / "celerybeat-schedule.db")
732# Cachalot: Database read cache.
733def _parse_cachalot_settings():
734 ttl = get_int_from_env("PAPERLESS_READ_CACHE_TTL", 3600)
735 ttl = min(ttl, 31536000) if ttl > 0 else 3600
736 _, redis_url = parse_redis_url(
737 os.getenv("PAPERLESS_READ_CACHE_REDIS_URL", _CHANNELS_REDIS_URL),
738 )
739 result = {
740 "CACHALOT_CACHE": "read-cache",
741 "CACHALOT_ENABLED": get_bool_from_env(
742 "PAPERLESS_DB_READ_CACHE_ENABLED",
743 default="no",
744 ),
745 "CACHALOT_FINAL_SQL_CHECK": True,
746 "CACHALOT_QUERY_KEYGEN": "paperless.db_cache.custom_get_query_cache_key",
747 "CACHALOT_TABLE_KEYGEN": "paperless.db_cache.custom_get_table_cache_key",
748 "CACHALOT_REDIS_URL": redis_url,
749 "CACHALOT_TIMEOUT": ttl,
750 }
751 return result
754cachalot_settings = _parse_cachalot_settings()
755CACHALOT_ENABLED = cachalot_settings["CACHALOT_ENABLED"]
756if CACHALOT_ENABLED: # pragma: no cover 756 ↛ 757line 756 didn't jump to line 757 because the condition on line 756 was never true
757 INSTALLED_APPS.append("cachalot")
758CACHALOT_CACHE = cachalot_settings["CACHALOT_CACHE"]
759CACHALOT_TIMEOUT = cachalot_settings["CACHALOT_TIMEOUT"]
760CACHALOT_QUERY_KEYGEN = cachalot_settings["CACHALOT_QUERY_KEYGEN"]
761CACHALOT_TABLE_KEYGEN = cachalot_settings["CACHALOT_TABLE_KEYGEN"]
762CACHALOT_FINAL_SQL_CHECK = cachalot_settings["CACHALOT_FINAL_SQL_CHECK"]
765# Django default & Cachalot cache configuration
766_CACHE_BACKEND = os.environ.get(
767 "PAPERLESS_CACHE_BACKEND",
768 "django.core.cache.backends.locmem.LocMemCache"
769 if DEBUG
770 else "django.core.cache.backends.redis.RedisCache",
771)
774def _parse_caches():
775 return {
776 "default": {
777 "BACKEND": _CACHE_BACKEND,
778 "LOCATION": _CHANNELS_REDIS_URL,
779 "KEY_PREFIX": _REDIS_KEY_PREFIX,
780 },
781 "read-cache": {
782 "BACKEND": _CACHE_BACKEND,
783 "LOCATION": cachalot_settings["CACHALOT_REDIS_URL"],
784 "KEY_PREFIX": _REDIS_KEY_PREFIX,
785 },
786 }
789CACHES = _parse_caches()
792def default_threads_per_worker(task_workers) -> int:
793 # always leave one core open
794 available_cores = max(multiprocessing.cpu_count(), 1)
795 try:
796 return max(math.floor(available_cores / task_workers), 1)
797 except NotImplementedError:
798 return 1
801THREADS_PER_WORKER = get_int_from_env(
802 "PAPERLESS_THREADS_PER_WORKER",
803 default_threads_per_worker(CELERY_WORKER_CONCURRENCY),
804)
806###############################################################################
807# Paperless Specific Settings #
808###############################################################################
810IGNORABLE_FILES: Final[list[str]] = [
811 ".DS_Store",
812 ".DS_STORE",
813 "._*",
814 ".stfolder/*",
815 ".stversions/*",
816 ".localized/*",
817 "desktop.ini",
818 "@eaDir/*",
819 "Thumbs.db",
820]
822CONSUMER_POLLING_INTERVAL = get_float_from_env("PAPERLESS_CONSUMER_POLLING_INTERVAL", 0)
824CONSUMER_STABILITY_DELAY = get_float_from_env("PAPERLESS_CONSUMER_STABILITY_DELAY", 5)
826CONSUMER_DELETE_DUPLICATES = get_bool_from_env("PAPERLESS_CONSUMER_DELETE_DUPLICATES")
828CONSUMER_RECURSIVE = get_bool_from_env("PAPERLESS_CONSUMER_RECURSIVE")
830# Ignore regex patterns, matched against filename only
831CONSUMER_IGNORE_PATTERNS = list(
832 json.loads(
833 os.getenv(
834 "PAPERLESS_CONSUMER_IGNORE_PATTERNS",
835 json.dumps([]),
836 ),
837 ),
838)
840# Directories to always ignore. These are matched by directory name, not full path
841CONSUMER_IGNORE_DIRS = list(
842 json.loads(
843 os.getenv(
844 "PAPERLESS_CONSUMER_IGNORE_DIRS",
845 json.dumps([]),
846 ),
847 ),
848)
850CONSUMER_SUBDIRS_AS_TAGS = get_bool_from_env("PAPERLESS_CONSUMER_SUBDIRS_AS_TAGS")
852CONSUMER_ENABLE_BARCODES: Final[bool] = get_bool_from_env(
853 "PAPERLESS_CONSUMER_ENABLE_BARCODES",
854)
856CONSUMER_BARCODE_TIFF_SUPPORT: Final[bool] = get_bool_from_env(
857 "PAPERLESS_CONSUMER_BARCODE_TIFF_SUPPORT",
858)
860CONSUMER_BARCODE_STRING: Final[str] = os.getenv(
861 "PAPERLESS_CONSUMER_BARCODE_STRING",
862 "PATCHT",
863)
865CONSUMER_ENABLE_ASN_BARCODE: Final[bool] = get_bool_from_env(
866 "PAPERLESS_CONSUMER_ENABLE_ASN_BARCODE",
867)
869CONSUMER_ASN_BARCODE_PREFIX: Final[str] = os.getenv(
870 "PAPERLESS_CONSUMER_ASN_BARCODE_PREFIX",
871 "ASN",
872)
874CONSUMER_BARCODE_UPSCALE: Final[float] = get_float_from_env(
875 "PAPERLESS_CONSUMER_BARCODE_UPSCALE",
876 0.0,
877)
879CONSUMER_BARCODE_DPI: Final[int] = get_int_from_env(
880 "PAPERLESS_CONSUMER_BARCODE_DPI",
881 300,
882)
884CONSUMER_BARCODE_MAX_PAGES: Final[int] = get_int_from_env(
885 "PAPERLESS_CONSUMER_BARCODE_MAX_PAGES",
886 0,
887)
889CONSUMER_BARCODE_RETAIN_SPLIT_PAGES = get_bool_from_env(
890 "PAPERLESS_CONSUMER_BARCODE_RETAIN_SPLIT_PAGES",
891)
893CONSUMER_ENABLE_TAG_BARCODE: Final[bool] = get_bool_from_env(
894 "PAPERLESS_CONSUMER_ENABLE_TAG_BARCODE",
895)
897CONSUMER_TAG_BARCODE_MAPPING = dict(
898 json.loads(
899 os.getenv(
900 "PAPERLESS_CONSUMER_TAG_BARCODE_MAPPING",
901 '{"TAG:(.*)": "\\\\g<1>"}',
902 ),
903 ),
904)
906CONSUMER_TAG_BARCODE_SPLIT: Final[bool] = get_bool_from_env(
907 "PAPERLESS_CONSUMER_TAG_BARCODE_SPLIT",
908)
910CONSUMER_STORE_BARCODE_VALUES: Final[bool] = get_bool_from_env(
911 "PAPERLESS_CONSUMER_STORE_BARCODE_VALUES",
912)
914CONSUMER_ENABLE_COLLATE_DOUBLE_SIDED: Final[bool] = get_bool_from_env(
915 "PAPERLESS_CONSUMER_ENABLE_COLLATE_DOUBLE_SIDED",
916)
918CONSUMER_COLLATE_DOUBLE_SIDED_SUBDIR_NAME: Final[str] = os.getenv(
919 "PAPERLESS_CONSUMER_COLLATE_DOUBLE_SIDED_SUBDIR_NAME",
920 "double-sided",
921)
923CONSUMER_COLLATE_DOUBLE_SIDED_TIFF_SUPPORT: Final[bool] = get_bool_from_env(
924 "PAPERLESS_CONSUMER_COLLATE_DOUBLE_SIDED_TIFF_SUPPORT",
925)
927CONSUMER_PDF_RECOVERABLE_MIME_TYPES = ("application/octet-stream",)
929OCR_PAGES = get_int_from_env("PAPERLESS_OCR_PAGES")
931# The default language that tesseract will attempt to use when parsing
932# documents. It should be a 3-letter language code consistent with ISO 639.
933OCR_LANGUAGE = os.getenv("PAPERLESS_OCR_LANGUAGE", "eng")
935# OCRmyPDF --output-type options are available.
936OCR_OUTPUT_TYPE = os.getenv("PAPERLESS_OCR_OUTPUT_TYPE", "pdfa")
938if os.environ.get("PAPERLESS_OCR_MODE", "") in ( 938 ↛ 942line 938 didn't jump to line 942 because the condition on line 938 was never true
939 "skip",
940 "skip_noarchive",
941): # pragma: no cover
942 OCR_MODE = "auto"
943else:
944 OCR_MODE = get_choice_from_env(
945 "PAPERLESS_OCR_MODE",
946 {"auto", "force", "redo", "off"},
947 default="auto",
948 )
950ARCHIVE_FILE_GENERATION = get_choice_from_env(
951 "PAPERLESS_ARCHIVE_FILE_GENERATION",
952 {"auto", "always", "never"},
953 default="auto",
954)
956OCR_IMAGE_DPI = get_int_from_env("PAPERLESS_OCR_IMAGE_DPI")
958OCR_CLEAN = os.getenv("PAPERLESS_OCR_CLEAN", "clean")
960OCR_DESKEW: Final[bool] = get_bool_from_env("PAPERLESS_OCR_DESKEW", "true")
962OCR_ROTATE_PAGES: Final[bool] = get_bool_from_env("PAPERLESS_OCR_ROTATE_PAGES", "true")
964OCR_ROTATE_PAGES_THRESHOLD: Final[float] = get_float_from_env(
965 "PAPERLESS_OCR_ROTATE_PAGES_THRESHOLD",
966 12.0,
967)
969OCR_MAX_IMAGE_PIXELS: Final[int | None] = get_int_from_env(
970 "PAPERLESS_OCR_MAX_IMAGE_PIXELS",
971)
973OCR_COLOR_CONVERSION_STRATEGY = os.getenv(
974 "PAPERLESS_OCR_COLOR_CONVERSION_STRATEGY",
975 "RGB",
976)
978OCR_USER_ARGS = os.getenv("PAPERLESS_OCR_USER_ARGS")
980MAX_IMAGE_PIXELS: Final[int | None] = get_int_from_env(
981 "PAPERLESS_MAX_IMAGE_PIXELS",
982)
984# GNUPG needs a home directory for some reason
985GNUPG_HOME = os.getenv("HOME", "/tmp")
987# Convert is part of the ImageMagick package
988CONVERT_BINARY = os.getenv("PAPERLESS_CONVERT_BINARY", "convert")
989CONVERT_TMPDIR = os.getenv("PAPERLESS_CONVERT_TMPDIR")
990CONVERT_MEMORY_LIMIT = os.getenv("PAPERLESS_CONVERT_MEMORY_LIMIT")
992GS_BINARY = os.getenv("PAPERLESS_GS_BINARY", "gs")
994# Fallback layout for .eml consumption
995EMAIL_PARSE_DEFAULT_LAYOUT = get_int_from_env(
996 "PAPERLESS_EMAIL_PARSE_DEFAULT_LAYOUT",
997 1, # MailRule.PdfLayout.TEXT_HTML but that can't be imported here
998)
1000# Trigger a script after every successful document consumption?
1001PRE_CONSUME_SCRIPT = os.getenv("PAPERLESS_PRE_CONSUME_SCRIPT")
1002POST_CONSUME_SCRIPT = os.getenv("PAPERLESS_POST_CONSUME_SCRIPT")
1004# Specify the default date order (for autodetected dates)
1005DATE_ORDER = os.getenv("PAPERLESS_DATE_ORDER", "DMY")
1006FILENAME_DATE_ORDER = os.getenv("PAPERLESS_FILENAME_DATE_ORDER")
1009# If not set, we will infer it at runtime
1010DATE_PARSER_LANGUAGES = (
1011 parse_dateparser_languages(
1012 os.getenv("PAPERLESS_DATE_PARSER_LANGUAGES"),
1013 )
1014 if os.getenv("PAPERLESS_DATE_PARSER_LANGUAGES")
1015 else None
1016)
1019# Maximum number of dates taken from document start to end to show as suggestions for
1020# `created` date in the frontend. Duplicates are removed, which can result in
1021# fewer dates shown.
1022NUMBER_OF_SUGGESTED_DATES = get_int_from_env("PAPERLESS_NUMBER_OF_SUGGESTED_DATES", 3)
1024# Specify the filename format for out files
1025FILENAME_FORMAT = os.getenv("PAPERLESS_FILENAME_FORMAT")
1027# If this is enabled, variables in filename format will resolve to
1028# empty-string instead of 'none'.
1029# Directories with 'empty names' are omitted, too.
1030FILENAME_FORMAT_REMOVE_NONE = get_bool_from_env(
1031 "PAPERLESS_FILENAME_FORMAT_REMOVE_NONE",
1032 "NO",
1033)
1035THUMBNAIL_FONT_NAME = os.getenv(
1036 "PAPERLESS_THUMBNAIL_FONT_NAME",
1037 "/usr/share/fonts/liberation/LiberationSerif-Regular.ttf",
1038)
1040# Tika settings
1041TIKA_ENABLED = get_bool_from_env("PAPERLESS_TIKA_ENABLED", "NO")
1042TIKA_ENDPOINT = os.getenv("PAPERLESS_TIKA_ENDPOINT", "http://localhost:9998")
1043TIKA_GOTENBERG_ENDPOINT = os.getenv(
1044 "PAPERLESS_TIKA_GOTENBERG_ENDPOINT",
1045 "http://localhost:3000",
1046)
1048# Tika parser is now integrated into the main parser registry
1049# No separate Django app needed
1051AUDIT_LOG_ENABLED = get_bool_from_env("PAPERLESS_AUDIT_LOG_ENABLED", "true")
1052if AUDIT_LOG_ENABLED: 1052 ↛ 1058line 1052 didn't jump to line 1058 because the condition on line 1052 was always true
1053 INSTALLED_APPS.append("auditlog")
1054 MIDDLEWARE.append("auditlog.middleware.AuditlogMiddleware")
1057# List dates that should be ignored when trying to parse date from document text
1058IGNORE_DATES: set[datetime.date] = set()
1060if os.getenv("PAPERLESS_IGNORE_DATES") is not None: 1060 ↛ 1061line 1060 didn't jump to line 1061 because the condition on line 1060 was never true
1061 IGNORE_DATES = parse_ignore_dates(os.getenv("PAPERLESS_IGNORE_DATES"), DATE_ORDER)
1063ENABLE_UPDATE_CHECK = os.getenv("PAPERLESS_ENABLE_UPDATE_CHECK", "default")
1064if ENABLE_UPDATE_CHECK != "default": 1064 ↛ 1065line 1064 didn't jump to line 1065 because the condition on line 1064 was never true
1065 ENABLE_UPDATE_CHECK = get_bool_from_env("PAPERLESS_ENABLE_UPDATE_CHECK")
1067APP_TITLE = os.getenv("PAPERLESS_APP_TITLE", None)
1068APP_LOGO = os.getenv("PAPERLESS_APP_LOGO", None)
1070###############################################################################
1071# Machine Learning #
1072###############################################################################
1075CLASSIFIER_LANGUAGES: Final[dict[str, str]] = {
1076 "dan": "danish",
1077 "nld": "dutch",
1078 "eng": "english",
1079 "fin": "finnish",
1080 "fra": "french",
1081 "deu": "german",
1082 "ita": "italian",
1083 "nor": "norwegian",
1084 "por": "portuguese",
1085 "rus": "russian",
1086 "spa": "spanish",
1087 "swe": "swedish",
1088}
1091def _get_llm_extra_params() -> dict[str, Any]:
1092 """
1093 Parse PAPERLESS_AI_LLM_EXTRA_PARAMS, a JSON object passed straight through
1094 to the LLM backend's request body.
1095 """
1096 raw = os.getenv("PAPERLESS_AI_LLM_EXTRA_PARAMS", "{}")
1097 try:
1098 parsed = json.loads(raw)
1099 except json.JSONDecodeError as e:
1100 raise ImproperlyConfigured(
1101 "PAPERLESS_AI_LLM_EXTRA_PARAMS must be valid JSON",
1102 ) from e
1103 if not isinstance(parsed, dict): 1103 ↛ 1104line 1103 didn't jump to line 1104 because the condition on line 1103 was never true
1104 raise ImproperlyConfigured(
1105 "PAPERLESS_AI_LLM_EXTRA_PARAMS must be a JSON object",
1106 )
1107 return parsed
1110def _get_classifier_language_setting(ocr_lang: str) -> str | None:
1111 """
1112 Maps the primary Tesseract language to the classifier's stemming
1113 language, or None if unsupported.
1115 Assumption: The primary language is first
1116 """
1117 return CLASSIFIER_LANGUAGES.get(ocr_lang.split("+", maxsplit=1)[0])
1120def _get_search_language_setting(ocr_lang: str) -> str | None:
1121 """
1122 Determine the Tantivy stemmer language.
1124 If PAPERLESS_SEARCH_LANGUAGE is explicitly set, it is validated against
1125 the languages supported by Tantivy's built-in stemmer and returned as-is.
1126 Otherwise the primary Tesseract language code from PAPERLESS_OCR_LANGUAGE
1127 is mapped to the corresponding ISO 639-1 code understood by Tantivy.
1128 Returns None when unset and the OCR language has no Tantivy stemmer.
1129 """
1130 explicit = os.environ.get("PAPERLESS_SEARCH_LANGUAGE")
1131 if explicit is not None: 1131 ↛ 1134line 1131 didn't jump to line 1134 because the condition on line 1131 was never true
1132 # Lazy import avoids any app-loading order concerns; _tokenizer has no
1133 # Django dependencies so this is safe.
1134 from documents.search._tokenizer import SUPPORTED_LANGUAGES
1136 return get_choice_from_env("PAPERLESS_SEARCH_LANGUAGE", SUPPORTED_LANGUAGES)
1138 # Infer from the primary Tesseract language code (ISO 639-2/T → ISO 639-1)
1139 primary = ocr_lang.split("+", maxsplit=1)[0].lower()
1140 _ocr_to_search: dict[str, str] = {
1141 "ara": "ar",
1142 "dan": "da",
1143 "nld": "nl",
1144 "eng": "en",
1145 "fin": "fi",
1146 "fra": "fr",
1147 "deu": "de",
1148 "ell": "el",
1149 "hun": "hu",
1150 "ita": "it",
1151 "nor": "no",
1152 "por": "pt",
1153 "ron": "ro",
1154 "rus": "ru",
1155 "spa": "es",
1156 "swe": "sv",
1157 "tam": "ta",
1158 "tur": "tr",
1159 }
1160 return _ocr_to_search.get(primary)
1163CLASSIFIER_LANGUAGE: str | None = _get_classifier_language_setting(OCR_LANGUAGE)
1165SEARCH_LANGUAGE: str | None = _get_search_language_setting(OCR_LANGUAGE)
1167###############################################################################
1168# Email Preprocessors #
1169###############################################################################
1171EMAIL_GNUPG_HOME: Final[str | None] = os.getenv("PAPERLESS_EMAIL_GNUPG_HOME")
1172EMAIL_ENABLE_GPG_DECRYPTOR: Final[bool] = get_bool_from_env(
1173 "PAPERLESS_ENABLE_GPG_DECRYPTOR",
1174)
1177###############################################################################
1178# Soft Delete #
1179###############################################################################
1180EMPTY_TRASH_DELAY = max(get_int_from_env("PAPERLESS_EMPTY_TRASH_DELAY", 30), 1)
1183###############################################################################
1184# Oauth Email #
1185###############################################################################
1186OAUTH_CALLBACK_BASE_URL = os.getenv("PAPERLESS_OAUTH_CALLBACK_BASE_URL")
1187GMAIL_OAUTH_CLIENT_ID = os.getenv("PAPERLESS_GMAIL_OAUTH_CLIENT_ID")
1188GMAIL_OAUTH_CLIENT_SECRET = os.getenv("PAPERLESS_GMAIL_OAUTH_CLIENT_SECRET")
1189GMAIL_OAUTH_ENABLED = bool(
1190 (OAUTH_CALLBACK_BASE_URL or PAPERLESS_URL)
1191 and GMAIL_OAUTH_CLIENT_ID
1192 and GMAIL_OAUTH_CLIENT_SECRET,
1193)
1194OUTLOOK_OAUTH_CLIENT_ID = os.getenv("PAPERLESS_OUTLOOK_OAUTH_CLIENT_ID")
1195OUTLOOK_OAUTH_CLIENT_SECRET = os.getenv("PAPERLESS_OUTLOOK_OAUTH_CLIENT_SECRET")
1196OUTLOOK_OAUTH_ENABLED = bool(
1197 (OAUTH_CALLBACK_BASE_URL or PAPERLESS_URL)
1198 and OUTLOOK_OAUTH_CLIENT_ID
1199 and OUTLOOK_OAUTH_CLIENT_SECRET,
1200)
1202###############################################################################
1203# Webhooks
1204###############################################################################
1205WEBHOOKS_ALLOWED_SCHEMES = {
1206 s.lower()
1207 for s in get_list_from_env(
1208 "PAPERLESS_WEBHOOKS_ALLOWED_SCHEMES",
1209 default=["http", "https"],
1210 )
1211}
1212WEBHOOKS_ALLOWED_PORTS = {
1213 int(p) for p in get_list_from_env("PAPERLESS_WEBHOOKS_ALLOWED_PORTS", default=[])
1214}
1215WEBHOOKS_ALLOW_INTERNAL_REQUESTS = get_bool_from_env(
1216 "PAPERLESS_WEBHOOKS_ALLOW_INTERNAL_REQUESTS",
1217 "true",
1218)
1220###############################################################################
1221# Remote Parser #
1222###############################################################################
1223REMOTE_OCR_ENGINE = os.getenv("PAPERLESS_REMOTE_OCR_ENGINE")
1224REMOTE_OCR_API_KEY = os.getenv("PAPERLESS_REMOTE_OCR_API_KEY")
1225REMOTE_OCR_ENDPOINT = os.getenv("PAPERLESS_REMOTE_OCR_ENDPOINT")
1226REMOTE_OCR_MODE = get_choice_from_env(
1227 "PAPERLESS_REMOTE_OCR_MODE",
1228 {"always", "workflow_only"},
1229 default="always",
1230)
1231REMOTE_OCR_ALLOW_INTERNAL_ENDPOINTS = get_bool_from_env(
1232 "PAPERLESS_REMOTE_OCR_ALLOW_INTERNAL_ENDPOINTS",
1233 "true",
1234)
1236################################################################################
1237# AI Settings #
1238################################################################################
1239AI_ENABLED = get_bool_from_env("PAPERLESS_AI_ENABLED", "NO")
1240LLM_EMBEDDING_BACKEND = get_choice_from_env(
1241 "PAPERLESS_AI_LLM_EMBEDDING_BACKEND",
1242 {"huggingface", "openai-like", "ollama"},
1243)
1244LLM_EMBEDDING_MODEL = os.getenv("PAPERLESS_AI_LLM_EMBEDDING_MODEL")
1245LLM_EMBEDDING_API_KEY = os.getenv("PAPERLESS_AI_LLM_EMBEDDING_API_KEY")
1246LLM_EMBEDDING_ENDPOINT = os.getenv("PAPERLESS_AI_LLM_EMBEDDING_ENDPOINT")
1247LLM_EMBEDDING_CHUNK_SIZE = get_int_from_env(
1248 "PAPERLESS_AI_LLM_EMBEDDING_CHUNK_SIZE",
1249 1024,
1250)
1251if LLM_EMBEDDING_CHUNK_SIZE < 1: 1251 ↛ 1252line 1251 didn't jump to line 1252 because the condition on line 1251 was never true
1252 raise ImproperlyConfigured("PAPERLESS_AI_LLM_EMBEDDING_CHUNK_SIZE must be >= 1")
1253LLM_CONTEXT_SIZE = get_int_from_env("PAPERLESS_AI_LLM_CONTEXT_SIZE", 8192)
1254if LLM_CONTEXT_SIZE < 1: 1254 ↛ 1255line 1254 didn't jump to line 1255 because the condition on line 1254 was never true
1255 raise ImproperlyConfigured("PAPERLESS_AI_LLM_CONTEXT_SIZE must be >= 1")
1256LLM_REQUEST_TIMEOUT = get_int_from_env("PAPERLESS_AI_LLM_REQUEST_TIMEOUT", 120)
1257if LLM_REQUEST_TIMEOUT < 1: 1257 ↛ 1258line 1257 didn't jump to line 1258 because the condition on line 1257 was never true
1258 raise ImproperlyConfigured("PAPERLESS_AI_LLM_REQUEST_TIMEOUT must be >= 1")
1259LLM_BACKEND = get_choice_from_env(
1260 "PAPERLESS_AI_LLM_BACKEND",
1261 {"ollama", "openai-like"},
1262)
1263LLM_MODEL = os.getenv("PAPERLESS_AI_LLM_MODEL")
1264LLM_API_KEY = os.getenv("PAPERLESS_AI_LLM_API_KEY")
1265LLM_ENDPOINT = os.getenv("PAPERLESS_AI_LLM_ENDPOINT")
1266LLM_OUTPUT_LANGUAGE = os.getenv("PAPERLESS_AI_LLM_OUTPUT_LANGUAGE")
1267LLM_ALLOW_INTERNAL_ENDPOINTS = get_bool_from_env(
1268 "PAPERLESS_AI_LLM_ALLOW_INTERNAL_ENDPOINTS",
1269 "true",
1270)
1271LLM_EXTRA_PARAMS = _get_llm_extra_params()