Coverage for documents/caching.py: 51%

100 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import hashlib 

4import logging 

5import uuid 

6from binascii import hexlify 

7from dataclasses import dataclass 

8from typing import TYPE_CHECKING 

9from typing import Final 

10 

11from django.core.cache import cache 

12 

13from documents.models import Document 

14 

15if TYPE_CHECKING: 15 ↛ 16line 15 didn't jump to line 16 because the condition on line 15 was never true

16 from documents.classifier import DocumentClassifier 

17 

18logger = logging.getLogger("paperless.caching") 

19 

20 

21@dataclass(frozen=True) 

22class MetadataCacheData: 

23 original_checksum: str 

24 original_metadata: list 

25 archive_checksum: str | None 

26 archive_metadata: list | None 

27 

28 

29@dataclass(frozen=True) 

30class SuggestionCacheData: 

31 classifier_version: int 

32 classifier_hash: str 

33 suggestions: dict 

34 

35 

36CLASSIFIER_VERSION_KEY: Final[str] = "classifier_version" 

37CLASSIFIER_HASH_KEY: Final[str] = "classifier_hash" 

38CLASSIFIER_MODIFIED_KEY: Final[str] = "classifier_modified" 

39# Marker distinguishing LLM suggestions from classifier-generated ones (whose 

40# FORMAT_VERSION lives in a much lower range - see DocumentClassifier). Bump 

41# this whenever cached suggestions must not be reused, including changes to 

42# their shape or interpretation, so a previous release's result cannot leak 

43# incompatible or obsolete behavior into the new one: 

44# 1000 - initial LLM suggestions cache (flat lists of resolved object ids 

45# per taxonomy field) 

46# 1001 - suggestions reshaped to {"existing_ids": [...], "new_names": 

47# [...]} per taxonomy field (#13676) 

48# 1002 - names are always generated and optional candidate mappings are 

49# validated separately, so candidate-anchored 1001 results are stale 

50LLM_CACHE_CLASSIFIER_VERSION: Final[int] = 1002 

51 

52CACHE_1_MINUTE: Final[int] = 60 

53CACHE_5_MINUTES: Final[int] = 5 * CACHE_1_MINUTE 

54CACHE_50_MINUTES: Final[int] = 50 * CACHE_1_MINUTE 

55# Deliberately longer than any entry it names 

56LLM_CACHE_GENERATION_TIMEOUT: Final[int] = 2 * CACHE_50_MINUTES 

57 

58 

59def get_suggestion_cache_key(document_id: int) -> str: 

60 """ 

61 Returns the basic key for a document's suggestions 

62 """ 

63 return f"doc_{document_id}_suggest" 

64 

65 

66def get_suggestion_cache(document_id: int) -> SuggestionCacheData | None: 

67 """ 

68 If possible, return the cached suggestions for the given document ID. 

69 The classifier needs to be matching in format and hash and the suggestions need to 

70 have been cached once. 

71 """ 

72 from documents.classifier import DocumentClassifier 

73 

74 doc_key = get_suggestion_cache_key(document_id) 

75 cache_hits = cache.get_many([CLASSIFIER_VERSION_KEY, CLASSIFIER_HASH_KEY, doc_key]) 

76 # The document suggestions are in the cache 

77 if doc_key in cache_hits: 

78 doc_suggestions: SuggestionCacheData = cache_hits[doc_key] 

79 # The classifier format is the same 

80 # The classifier hash is the same 

81 # Then the suggestions can be used 

82 if ( 

83 CLASSIFIER_VERSION_KEY in cache_hits 

84 and cache_hits[CLASSIFIER_VERSION_KEY] == DocumentClassifier.FORMAT_VERSION 

85 and cache_hits[CLASSIFIER_VERSION_KEY] == doc_suggestions.classifier_version 

86 ) and ( 

87 CLASSIFIER_HASH_KEY in cache_hits 

88 and cache_hits[CLASSIFIER_HASH_KEY] == doc_suggestions.classifier_hash 

89 ): 

90 return doc_suggestions 

91 else: # pragma: no cover 

92 # Remove the key because something didn't match 

93 cache.delete(doc_key) 

94 return None 

95 

96 

97def set_suggestions_cache( 

98 document_id: int, 

99 suggestions: dict, 

100 classifier: DocumentClassifier | None, 

101 *, 

102 timeout=CACHE_50_MINUTES, 

103) -> None: 

104 """ 

105 Caches the given suggestions, which were generated by the given classifier. If there is no classifier, 

106 this function is a no-op (there won't be suggestions then anyway) 

107 """ 

108 if classifier is not None: 

109 doc_key = get_suggestion_cache_key(document_id) 

110 cache.set( 

111 doc_key, 

112 SuggestionCacheData( 

113 classifier.FORMAT_VERSION, 

114 hexlify(classifier.last_auto_type_hash).decode(), 

115 suggestions, 

116 ), 

117 timeout, 

118 ) 

119 

120 

121def refresh_suggestions_cache( 

122 document_id: int, 

123 *, 

124 timeout: int = CACHE_50_MINUTES, 

125) -> None: 

126 """ 

127 Refreshes the expiration of the suggestions for the given document ID 

128 to the given timeout 

129 """ 

130 doc_key = get_suggestion_cache_key(document_id) 

131 cache.touch(doc_key, timeout) 

132 

133 

134def invalidate_suggestions_cache(document_id: int) -> None: 

135 """Invalidate classifier-generated suggestions for a document.""" 

136 cache.delete(get_suggestion_cache_key(document_id)) 

137 

138 

139def _llm_generation_key(document_id: int) -> str: 

140 return f"{get_suggestion_cache_key(document_id)}_llm_generation" 

141 

142 

143def _llm_variant_key(document_id: int, backend: str) -> str: 

144 """Cache key for one LLM configuration and permission scope. 

145 

146 ``backend`` identifies the variant - model, endpoint, output language and 

147 requesting user. 

148 

149 Generating the token on first use lets invalidate_llm_suggestions_cache() 

150 be no-op for documents that never had AI suggestions. 

151 """ 

152 generation_key = _llm_generation_key(document_id) 

153 generation = cache.get_or_set( 

154 generation_key, 

155 lambda: uuid.uuid4().hex, 

156 timeout=LLM_CACHE_GENERATION_TIMEOUT, 

157 ) 

158 cache.touch(generation_key, LLM_CACHE_GENERATION_TIMEOUT) 

159 backend_hash = hashlib.sha256(backend.encode()).hexdigest()[:16] 

160 return f"{get_suggestion_cache_key(document_id)}_llm_{generation}_{backend_hash}" 

161 

162 

163def get_llm_suggestion_cache( 

164 document_id: int, 

165 backend: str, 

166) -> SuggestionCacheData | None: 

167 data: SuggestionCacheData = cache.get(_llm_variant_key(document_id, backend)) 

168 

169 if ( 

170 data 

171 and data.classifier_version == LLM_CACHE_CLASSIFIER_VERSION 

172 and data.classifier_hash == backend 

173 ): 

174 return data 

175 

176 return None 

177 

178 

179def set_llm_suggestions_cache( 

180 document_id: int, 

181 suggestions: dict, 

182 *, 

183 backend: str, 

184 timeout: int = CACHE_50_MINUTES, 

185) -> None: 

186 """ 

187 Cache LLM-generated suggestions using a backend-specific identifier 

188 (e.g. 'openai-like:gpt-4'). 

189 """ 

190 cache.set( 

191 _llm_variant_key(document_id, backend), 

192 SuggestionCacheData( 

193 classifier_version=LLM_CACHE_CLASSIFIER_VERSION, 

194 classifier_hash=backend, 

195 suggestions=suggestions, 

196 ), 

197 timeout, 

198 ) 

199 

200 

201def refresh_llm_suggestions_cache( 

202 document_id: int, 

203 backend: str, 

204 *, 

205 timeout: int = CACHE_50_MINUTES, 

206) -> None: 

207 """ 

208 Refreshes the expiration of one cached LLM suggestion variant. 

209 """ 

210 cache.touch(_llm_variant_key(document_id, backend), timeout) 

211 

212 

213def invalidate_llm_suggestions_cache( 

214 document_id: int, 

215) -> None: 

216 """ 

217 Invalidate every LLM suggestion variant for a document. 

218 """ 

219 generation_key = _llm_generation_key(document_id) 

220 if cache.get(generation_key) is not None: 220 ↛ 221line 220 didn't jump to line 221 because the condition on line 220 was never true

221 cache.set( 

222 generation_key, 

223 uuid.uuid4().hex, 

224 timeout=LLM_CACHE_GENERATION_TIMEOUT, 

225 ) 

226 

227 

228def get_metadata_cache_key(document_id: int) -> str: 

229 """ 

230 Returns the basic key for a document's metadata 

231 """ 

232 return f"doc_{document_id}_metadata" 

233 

234 

235def get_metadata_cache(document_id: int) -> MetadataCacheData | None: 

236 """ 

237 Returns the cached document metadata for the given document ID, as long as the metadata 

238 was cached once and the checksums have not changed 

239 """ 

240 doc_key = get_metadata_cache_key(document_id) 

241 doc_metadata: MetadataCacheData | None = cache.get(doc_key) 

242 # The metadata exists in the cache 

243 if doc_metadata is not None: 

244 try: 

245 doc = Document.objects.only( 

246 "pk", 

247 "checksum", 

248 "archive_checksum", 

249 "archive_filename", 

250 ).get(pk=document_id) 

251 # The original checksums match 

252 # If it has one, the archive checksums match 

253 # Then, we can use the metadata 

254 if ( 

255 doc_metadata.original_checksum == doc.checksum 

256 and doc.has_archive_version 

257 and doc_metadata.archive_checksum is not None 

258 and doc_metadata.archive_checksum == doc.archive_checksum 

259 ): 

260 # Refresh cache 

261 cache.touch(doc_key, CACHE_50_MINUTES) 

262 return doc_metadata 

263 else: # pragma: no cover 

264 # Something didn't match, delete the key 

265 cache.delete(doc_key) 

266 except Document.DoesNotExist: # pragma: no cover 

267 # Basically impossible, but the key existed, but the Document didn't 

268 cache.delete(doc_key) 

269 return None 

270 

271 

272def set_metadata_cache( 

273 document: Document, 

274 original_metadata: list, 

275 archive_metadata: list | None, 

276 *, 

277 timeout=CACHE_50_MINUTES, 

278) -> None: 

279 """ 

280 Sets the metadata into cache for the given Document 

281 """ 

282 doc_key = get_metadata_cache_key(document.pk) 

283 cache.set( 

284 doc_key, 

285 MetadataCacheData( 

286 document.checksum, 

287 original_metadata, 

288 document.archive_checksum, 

289 archive_metadata, 

290 ), 

291 timeout, 

292 ) 

293 

294 

295def refresh_metadata_cache( 

296 document_id: int, 

297 *, 

298 timeout: int = CACHE_50_MINUTES, 

299) -> None: 

300 """ 

301 Refreshes the expiration of the metadata for the given document ID 

302 to the given timeout 

303 """ 

304 doc_key = get_metadata_cache_key(document_id) 

305 cache.touch(doc_key, timeout) 

306 

307 

308def get_thumbnail_modified_key(document_id: int) -> str: 

309 """ 

310 Builds the key to store a thumbnail's timestamp 

311 """ 

312 return f"doc_{document_id}_thumbnail_modified" 

313 

314 

315def clear_document_caches(document_id: int) -> None: 

316 """ 

317 Removes all cached items for the given document 

318 """ 

319 cache.delete_many( 

320 [ 

321 get_suggestion_cache_key(document_id), 

322 get_metadata_cache_key(document_id), 

323 get_thumbnail_modified_key(document_id), 

324 ], 

325 ) 

326 invalidate_llm_suggestions_cache(document_id)