Coverage for paperless/settings/custom.py: 50%

90 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1import datetime 

2import logging 

3import os 

4from hashlib import sha256 

5from pathlib import Path 

6from typing import Any 

7 

8from celery.schedules import crontab 

9from dateparser.languages.loader import LocaleDataLoader 

10 

11from paperless.settings.parsers import get_choice_from_env 

12from paperless.settings.parsers import get_int_from_env 

13from paperless.settings.parsers import parse_dict_from_str 

14 

15logger = logging.getLogger(__name__) 

16 

17 

18def parse_hosting_settings() -> tuple[str | None, str, str, str, str]: 

19 script_name = os.getenv("PAPERLESS_FORCE_SCRIPT_NAME") 

20 base_url = (script_name or "") + "/" 

21 login_url = base_url + "accounts/login/" 

22 login_redirect_url = base_url + "dashboard" 

23 logout_redirect_url = os.getenv( 

24 "PAPERLESS_LOGOUT_REDIRECT_URL", 

25 login_url + "?loggedout=1", 

26 ) 

27 return script_name, base_url, login_url, login_redirect_url, logout_redirect_url 

28 

29 

30def parse_redis_url(env_redis: str | None) -> tuple[str, str]: 

31 """ 

32 Gets the Redis information from the environment or a default and handles 

33 converting from incompatible django_channels and celery formats. 

34 

35 Returns a tuple of (celery_url, channels_url) 

36 """ 

37 

38 # Not set, return a compatible default 

39 if env_redis is None: 39 ↛ 40line 39 didn't jump to line 40 because the condition on line 39 was never true

40 return ("redis://localhost:6379", "redis://localhost:6379") 

41 

42 if "unix" in env_redis.lower(): 42 ↛ 45line 42 didn't jump to line 45 because the condition on line 42 was never true

43 # channels_redis socket format, looks like: 

44 # "unix:///path/to/redis.sock" 

45 _, path = env_redis.split(":", maxsplit=1) 

46 # Optionally setting a db number 

47 if "?db=" in env_redis: 

48 path, number = path.split("?db=") 

49 return (f"redis+socket:{path}?virtual_host={number}", env_redis) 

50 else: 

51 return (f"redis+socket:{path}", env_redis) 

52 

53 elif "+socket" in env_redis.lower(): 53 ↛ 56line 53 didn't jump to line 56 because the condition on line 53 was never true

54 # celery socket style, looks like: 

55 # "redis+socket:///path/to/redis.sock" 

56 _, path = env_redis.split(":", maxsplit=1) 

57 if "?virtual_host=" in env_redis: 

58 # Virtual host (aka db number) 

59 path, number = path.split("?virtual_host=") 

60 return (env_redis, f"unix:{path}?db={number}") 

61 else: 

62 return (env_redis, f"unix:{path}") 

63 

64 # Not a socket 

65 return (env_redis, env_redis) 

66 

67 

68def parse_beat_schedule() -> dict: 

69 """ 

70 Configures the scheduled tasks, according to default or 

71 environment variables. Task expiration is configured so the task will 

72 expire (and not run), shortly before the default frequency will put another 

73 of the same task into the queue 

74 

75 

76 https://docs.celeryq.dev/en/stable/userguide/periodic-tasks.html#beat-entries 

77 https://docs.celeryq.dev/en/latest/userguide/calling.html#expiration 

78 """ 

79 schedule = {} 

80 tasks = [ 

81 { 

82 "name": "Check all e-mail accounts", 

83 "env_key": "PAPERLESS_EMAIL_TASK_CRON", 

84 # Default every ten minutes 

85 "env_default": "*/10 * * * *", 

86 "task": "paperless_mail.tasks.process_mail_accounts", 

87 "options": { 

88 # 1 minute before default schedule sends again 

89 "expires": 9.0 * 60.0, 

90 }, 

91 }, 

92 { 

93 "name": "Train the classifier", 

94 "env_key": "PAPERLESS_TRAIN_TASK_CRON", 

95 # Default hourly at 5 minutes past the hour 

96 "env_default": "5 */1 * * *", 

97 "task": "documents.tasks.train_classifier", 

98 "options": { 

99 # 1 minute before default schedule sends again 

100 "expires": 59.0 * 60.0, 

101 }, 

102 }, 

103 { 

104 "name": "Optimize the index", 

105 "env_key": "PAPERLESS_INDEX_TASK_CRON", 

106 # Default daily at midnight 

107 "env_default": "0 0 * * *", 

108 "task": "documents.tasks.index_optimize", 

109 "options": { 

110 # 1 hour before default schedule sends again 

111 "expires": 23.0 * 60.0 * 60.0, 

112 }, 

113 }, 

114 { 

115 "name": "Perform sanity check", 

116 "env_key": "PAPERLESS_SANITY_TASK_CRON", 

117 # Default Sunday at 00:30 

118 "env_default": "30 0 * * sun", 

119 "task": "documents.tasks.sanity_check", 

120 "options": { 

121 # 1 hour before default schedule sends again 

122 "expires": ((7.0 * 24.0) - 1.0) * 60.0 * 60.0, 

123 }, 

124 }, 

125 { 

126 "name": "Empty trash", 

127 "env_key": "PAPERLESS_EMPTY_TRASH_TASK_CRON", 

128 # Default daily at 01:00 

129 "env_default": "0 1 * * *", 

130 "task": "documents.tasks.empty_trash", 

131 "options": { 

132 # 1 hour before default schedule sends again 

133 "expires": 23.0 * 60.0 * 60.0, 

134 }, 

135 }, 

136 { 

137 "name": "Check and run scheduled workflows", 

138 "env_key": "PAPERLESS_WORKFLOW_SCHEDULED_TASK_CRON", 

139 # Default hourly at 5 minutes past the hour 

140 "env_default": "5 */1 * * *", 

141 "task": "documents.tasks.check_scheduled_workflows", 

142 "options": { 

143 # 1 minute before default schedule sends again 

144 "expires": 59.0 * 60.0, 

145 }, 

146 }, 

147 { 

148 "name": "Rebuild LLM index", 

149 "env_key": "PAPERLESS_LLM_INDEX_TASK_CRON", 

150 # Default daily at 02:10 

151 "env_default": "10 2 * * *", 

152 "task": "documents.tasks.llmindex_index", 

153 "options": { 

154 # 1 hour before default schedule sends again 

155 "expires": 23.0 * 60.0 * 60.0, 

156 }, 

157 }, 

158 { 

159 "name": "Cleanup expired share link bundles", 

160 "env_key": "PAPERLESS_SHARE_LINK_BUNDLE_CLEANUP_CRON", 

161 # Default daily at 02:00 

162 "env_default": "0 2 * * *", 

163 "task": "documents.tasks.cleanup_expired_share_link_bundles", 

164 "options": { 

165 # 1 hour before default schedule sends again 

166 "expires": 23.0 * 60.0 * 60.0, 

167 }, 

168 }, 

169 ] 

170 for task in tasks: 

171 # Either get the environment setting or use the default 

172 value = os.getenv(task["env_key"], task["env_default"]) 

173 # Don't add disabled tasks to the schedule 

174 if value == "disable": 174 ↛ 175line 174 didn't jump to line 175 because the condition on line 174 was never true

175 continue 

176 if ( 

177 task["env_key"] == "PAPERLESS_EMAIL_TASK_CRON" 

178 and task["env_key"] not in os.environ 

179 ): 

180 # Spread default polling across the ten-minute interval. 

181 secret = os.environ["PAPERLESS_SECRET_KEY"].encode() 

182 offset = int.from_bytes(sha256(secret).digest()) % 10 

183 minutes = ",".join(str(minute) for minute in range(offset, 60, 10)) 

184 value = f"{minutes} * * * *" 

185 # I find https://crontab.guru/ super helpful 

186 # crontab(5) format 

187 # - five time-and-date fields 

188 # - separated by at least one blank 

189 minute, hour, day_month, month, day_week = value.split(" ") 

190 

191 schedule[task["name"]] = { 

192 "task": task["task"], 

193 "schedule": crontab(minute, hour, day_week, day_month, month), 

194 "options": { 

195 **task["options"], 

196 # PaperlessTask.TriggerSource.SCHEDULED -- models can't be imported here 

197 "headers": {"trigger_source": "scheduled"}, 

198 }, 

199 } 

200 

201 return schedule 

202 

203 

204def parse_db_settings(data_dir: Path) -> dict[str, dict[str, Any]]: 

205 """Parse database settings from environment variables. 

206 

207 Core connection variables (no deprecation): 

208 - PAPERLESS_DBENGINE (sqlite/postgresql/mariadb) 

209 - PAPERLESS_DBHOST, PAPERLESS_DBPORT 

210 - PAPERLESS_DBNAME, PAPERLESS_DBUSER, PAPERLESS_DBPASS 

211 

212 Advanced options can be set via: 

213 - Legacy individual env vars (deprecated in v3.0, removed in v3.2) 

214 - PAPERLESS_DB_OPTIONS (recommended v3+ approach) 

215 

216 Args: 

217 data_dir: The data directory path for SQLite database location. 

218 

219 Returns: 

220 A databases dict suitable for Django DATABASES setting. 

221 """ 

222 engine = get_choice_from_env( 

223 "PAPERLESS_DBENGINE", 

224 {"sqlite", "postgresql", "mariadb"}, 

225 ) 

226 if engine is None: 226 ↛ 234line 226 didn't jump to line 234 because the condition on line 226 was always true

227 # MariaDB users already had to set PAPERLESS_DBENGINE, so it was picked up above 

228 # SQLite users didn't need to set anything 

229 engine = "postgresql" if "PAPERLESS_DBHOST" in os.environ else "sqlite" 

230 

231 db_config: dict[str, Any] 

232 base_options: dict[str, Any] 

233 

234 match engine: 

235 case "sqlite": 235 ↛ 258line 235 didn't jump to line 258 because the pattern on line 235 always matched

236 db_config = { 

237 "ENGINE": "django.db.backends.sqlite3", 

238 "NAME": str((data_dir / "db.sqlite3").resolve()), 

239 } 

240 base_options = { 

241 # Django splits init_command on ";" and calls conn.execute() 

242 # once per statement, so multiple PRAGMAs work correctly. 

243 # foreign_keys is omitted — Django sets it natively. 

244 "init_command": ( 

245 "PRAGMA journal_mode=WAL;" 

246 "PRAGMA synchronous=NORMAL;" 

247 "PRAGMA busy_timeout=5000;" 

248 "PRAGMA temp_store=MEMORY;" 

249 "PRAGMA mmap_size=134217728;" 

250 "PRAGMA journal_size_limit=67108864;" 

251 "PRAGMA cache_size=-8000" # negative = KiB; -8000 ≈ 8 MB 

252 ), 

253 # IMMEDIATE acquires the write lock at BEGIN, ensuring 

254 # busy_timeout is respected from the start of the transaction. 

255 "transaction_mode": "IMMEDIATE", 

256 } 

257 

258 case "postgresql": 

259 db_config = { 

260 "ENGINE": "django.db.backends.postgresql", 

261 "HOST": os.getenv("PAPERLESS_DBHOST"), 

262 "NAME": os.getenv("PAPERLESS_DBNAME", "paperless"), 

263 "USER": os.getenv("PAPERLESS_DBUSER", "paperless"), 

264 "PASSWORD": os.getenv("PAPERLESS_DBPASS", "paperless"), 

265 # Validate pooled connections so a connection closed server-side 

266 # is replaced rather than handed out as "the connection is closed". 

267 "CONN_HEALTH_CHECKS": True, 

268 } 

269 

270 base_options = { 

271 "sslmode": os.getenv("PAPERLESS_DBSSLMODE", "prefer"), 

272 "sslrootcert": os.getenv("PAPERLESS_DBSSLROOTCERT"), 

273 "sslcert": os.getenv("PAPERLESS_DBSSLCERT"), 

274 "sslkey": os.getenv("PAPERLESS_DBSSLKEY"), 

275 "application_name": "paperless-ngx", 

276 } 

277 

278 if (pool_size := get_int_from_env("PAPERLESS_DB_POOLSIZE")) is not None: 

279 base_options["pool"] = { 

280 "min_size": 1, 

281 "max_size": pool_size, 

282 } 

283 

284 case "mariadb": 

285 db_config = { 

286 "ENGINE": "django.db.backends.mysql", 

287 "HOST": os.getenv("PAPERLESS_DBHOST"), 

288 "NAME": os.getenv("PAPERLESS_DBNAME", "paperless"), 

289 "USER": os.getenv("PAPERLESS_DBUSER", "paperless"), 

290 "PASSWORD": os.getenv("PAPERLESS_DBPASS", "paperless"), 

291 } 

292 

293 base_options = { 

294 "read_default_file": "/etc/mysql/my.cnf", 

295 "charset": "utf8mb4", 

296 "collation": "utf8mb4_unicode_ci", 

297 "ssl_mode": os.getenv("PAPERLESS_DBSSLMODE", "PREFERRED"), 

298 "ssl": { 

299 "ca": os.getenv("PAPERLESS_DBSSLROOTCERT"), 

300 "cert": os.getenv("PAPERLESS_DBSSLCERT"), 

301 "key": os.getenv("PAPERLESS_DBSSLKEY"), 

302 }, 

303 # READ COMMITTED eliminates gap locking and reduces deadlocks. 

304 # Django also defaults to "read committed" for MySQL/MariaDB, but 

305 # we set it explicitly so the intent is clear and survives any 

306 # future changes to Django's default. 

307 # Requires binlog_format=ROW if binary logging is enabled. 

308 "isolation_level": "read committed", 

309 } 

310 case _: # pragma: no cover 

311 raise NotImplementedError(engine) 

312 

313 # Handle port setting for external databases 

314 if ( 314 ↛ 318line 314 didn't jump to line 318 because the condition on line 314 was never true

315 engine in ("postgresql", "mariadb") 

316 and (port := get_int_from_env("PAPERLESS_DBPORT")) is not None 

317 ): 

318 db_config["PORT"] = port 

319 

320 # Handle timeout setting (common across all engines, different key names) 

321 if (timeout := get_int_from_env("PAPERLESS_DB_TIMEOUT")) is not None: 321 ↛ 322line 321 didn't jump to line 322 because the condition on line 321 was never true

322 timeout_key = "timeout" if engine == "sqlite" else "connect_timeout" 

323 base_options[timeout_key] = timeout 

324 

325 # Apply PAPERLESS_DB_OPTIONS overrides 

326 db_config["OPTIONS"] = parse_dict_from_str( 

327 os.getenv("PAPERLESS_DB_OPTIONS"), 

328 defaults=base_options, 

329 separator=",", 

330 type_map={ 

331 # SQLite options 

332 "timeout": int, 

333 # Postgres/MariaDB options 

334 "connect_timeout": int, 

335 "pool.min_size": int, 

336 "pool.max_size": int, 

337 }, 

338 ) 

339 

340 return {"default": db_config} 

341 

342 

343def parse_dateparser_languages(languages: str | None) -> list[str]: 

344 language_list = languages.split("+") if languages else [] 

345 # There is an unfixed issue in zh-Hant and zh-Hans locales in the dateparser lib. 

346 # See: https://github.com/scrapinghub/dateparser/issues/875 

347 for index, language in enumerate(language_list): 

348 if language.startswith("zh-") and "zh" not in language_list: 

349 logger.warning( 

350 f"Chinese locale detected: {language}. dateparser might fail to parse" 

351 f' some dates with this locale, so Chinese ("zh") will be used as a fallback.', 

352 ) 

353 language_list.append("zh") 

354 

355 return list(LocaleDataLoader().get_locale_map(locales=language_list)) 

356 

357 

358def parse_ignore_dates( 

359 env_ignore: str, 

360 date_order: str, 

361) -> set[datetime.date]: 

362 """ 

363 If the PAPERLESS_IGNORE_DATES environment variable is set, parse the 

364 user provided string(s) into dates 

365 

366 Args: 

367 env_ignore (str): The value of the environment variable, comma separated dates 

368 date_order (str): The format of the date strings. 

369 

370 Returns: 

371 set[datetime.date]: The set of parsed date objects 

372 """ 

373 import dateparser 

374 

375 ignored_dates = set() 

376 for s in env_ignore.split(","): 

377 d = dateparser.parse( 

378 s, 

379 settings={ 

380 "DATE_ORDER": date_order, 

381 }, 

382 ) 

383 if d: 

384 ignored_dates.add(d.date()) 

385 return ignored_dates