Coverage for documents/caching.py: 51%
100 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import hashlib
4import logging
5import uuid
6from binascii import hexlify
7from dataclasses import dataclass
8from typing import TYPE_CHECKING
9from typing import Final
11from django.core.cache import cache
13from documents.models import Document
15if TYPE_CHECKING: 15 ↛ 16line 15 didn't jump to line 16 because the condition on line 15 was never true
16 from documents.classifier import DocumentClassifier
18logger = logging.getLogger("paperless.caching")
21@dataclass(frozen=True)
22class MetadataCacheData:
23 original_checksum: str
24 original_metadata: list
25 archive_checksum: str | None
26 archive_metadata: list | None
29@dataclass(frozen=True)
30class SuggestionCacheData:
31 classifier_version: int
32 classifier_hash: str
33 suggestions: dict
36CLASSIFIER_VERSION_KEY: Final[str] = "classifier_version"
37CLASSIFIER_HASH_KEY: Final[str] = "classifier_hash"
38CLASSIFIER_MODIFIED_KEY: Final[str] = "classifier_modified"
39# Marker distinguishing LLM suggestions from classifier-generated ones (whose
40# FORMAT_VERSION lives in a much lower range - see DocumentClassifier). Bump
41# this whenever cached suggestions must not be reused, including changes to
42# their shape or interpretation, so a previous release's result cannot leak
43# incompatible or obsolete behavior into the new one:
44# 1000 - initial LLM suggestions cache (flat lists of resolved object ids
45# per taxonomy field)
46# 1001 - suggestions reshaped to {"existing_ids": [...], "new_names":
47# [...]} per taxonomy field (#13676)
48# 1002 - names are always generated and optional candidate mappings are
49# validated separately, so candidate-anchored 1001 results are stale
50LLM_CACHE_CLASSIFIER_VERSION: Final[int] = 1002
52CACHE_1_MINUTE: Final[int] = 60
53CACHE_5_MINUTES: Final[int] = 5 * CACHE_1_MINUTE
54CACHE_50_MINUTES: Final[int] = 50 * CACHE_1_MINUTE
55# Deliberately longer than any entry it names
56LLM_CACHE_GENERATION_TIMEOUT: Final[int] = 2 * CACHE_50_MINUTES
59def get_suggestion_cache_key(document_id: int) -> str:
60 """
61 Returns the basic key for a document's suggestions
62 """
63 return f"doc_{document_id}_suggest"
66def get_suggestion_cache(document_id: int) -> SuggestionCacheData | None:
67 """
68 If possible, return the cached suggestions for the given document ID.
69 The classifier needs to be matching in format and hash and the suggestions need to
70 have been cached once.
71 """
72 from documents.classifier import DocumentClassifier
74 doc_key = get_suggestion_cache_key(document_id)
75 cache_hits = cache.get_many([CLASSIFIER_VERSION_KEY, CLASSIFIER_HASH_KEY, doc_key])
76 # The document suggestions are in the cache
77 if doc_key in cache_hits:
78 doc_suggestions: SuggestionCacheData = cache_hits[doc_key]
79 # The classifier format is the same
80 # The classifier hash is the same
81 # Then the suggestions can be used
82 if (
83 CLASSIFIER_VERSION_KEY in cache_hits
84 and cache_hits[CLASSIFIER_VERSION_KEY] == DocumentClassifier.FORMAT_VERSION
85 and cache_hits[CLASSIFIER_VERSION_KEY] == doc_suggestions.classifier_version
86 ) and (
87 CLASSIFIER_HASH_KEY in cache_hits
88 and cache_hits[CLASSIFIER_HASH_KEY] == doc_suggestions.classifier_hash
89 ):
90 return doc_suggestions
91 else: # pragma: no cover
92 # Remove the key because something didn't match
93 cache.delete(doc_key)
94 return None
97def set_suggestions_cache(
98 document_id: int,
99 suggestions: dict,
100 classifier: DocumentClassifier | None,
101 *,
102 timeout=CACHE_50_MINUTES,
103) -> None:
104 """
105 Caches the given suggestions, which were generated by the given classifier. If there is no classifier,
106 this function is a no-op (there won't be suggestions then anyway)
107 """
108 if classifier is not None:
109 doc_key = get_suggestion_cache_key(document_id)
110 cache.set(
111 doc_key,
112 SuggestionCacheData(
113 classifier.FORMAT_VERSION,
114 hexlify(classifier.last_auto_type_hash).decode(),
115 suggestions,
116 ),
117 timeout,
118 )
121def refresh_suggestions_cache(
122 document_id: int,
123 *,
124 timeout: int = CACHE_50_MINUTES,
125) -> None:
126 """
127 Refreshes the expiration of the suggestions for the given document ID
128 to the given timeout
129 """
130 doc_key = get_suggestion_cache_key(document_id)
131 cache.touch(doc_key, timeout)
134def invalidate_suggestions_cache(document_id: int) -> None:
135 """Invalidate classifier-generated suggestions for a document."""
136 cache.delete(get_suggestion_cache_key(document_id))
139def _llm_generation_key(document_id: int) -> str:
140 return f"{get_suggestion_cache_key(document_id)}_llm_generation"
143def _llm_variant_key(document_id: int, backend: str) -> str:
144 """Cache key for one LLM configuration and permission scope.
146 ``backend`` identifies the variant - model, endpoint, output language and
147 requesting user.
149 Generating the token on first use lets invalidate_llm_suggestions_cache()
150 be no-op for documents that never had AI suggestions.
151 """
152 generation_key = _llm_generation_key(document_id)
153 generation = cache.get_or_set(
154 generation_key,
155 lambda: uuid.uuid4().hex,
156 timeout=LLM_CACHE_GENERATION_TIMEOUT,
157 )
158 cache.touch(generation_key, LLM_CACHE_GENERATION_TIMEOUT)
159 backend_hash = hashlib.sha256(backend.encode()).hexdigest()[:16]
160 return f"{get_suggestion_cache_key(document_id)}_llm_{generation}_{backend_hash}"
163def get_llm_suggestion_cache(
164 document_id: int,
165 backend: str,
166) -> SuggestionCacheData | None:
167 data: SuggestionCacheData = cache.get(_llm_variant_key(document_id, backend))
169 if (
170 data
171 and data.classifier_version == LLM_CACHE_CLASSIFIER_VERSION
172 and data.classifier_hash == backend
173 ):
174 return data
176 return None
179def set_llm_suggestions_cache(
180 document_id: int,
181 suggestions: dict,
182 *,
183 backend: str,
184 timeout: int = CACHE_50_MINUTES,
185) -> None:
186 """
187 Cache LLM-generated suggestions using a backend-specific identifier
188 (e.g. 'openai-like:gpt-4').
189 """
190 cache.set(
191 _llm_variant_key(document_id, backend),
192 SuggestionCacheData(
193 classifier_version=LLM_CACHE_CLASSIFIER_VERSION,
194 classifier_hash=backend,
195 suggestions=suggestions,
196 ),
197 timeout,
198 )
201def refresh_llm_suggestions_cache(
202 document_id: int,
203 backend: str,
204 *,
205 timeout: int = CACHE_50_MINUTES,
206) -> None:
207 """
208 Refreshes the expiration of one cached LLM suggestion variant.
209 """
210 cache.touch(_llm_variant_key(document_id, backend), timeout)
213def invalidate_llm_suggestions_cache(
214 document_id: int,
215) -> None:
216 """
217 Invalidate every LLM suggestion variant for a document.
218 """
219 generation_key = _llm_generation_key(document_id)
220 if cache.get(generation_key) is not None: 220 ↛ 221line 220 didn't jump to line 221 because the condition on line 220 was never true
221 cache.set(
222 generation_key,
223 uuid.uuid4().hex,
224 timeout=LLM_CACHE_GENERATION_TIMEOUT,
225 )
228def get_metadata_cache_key(document_id: int) -> str:
229 """
230 Returns the basic key for a document's metadata
231 """
232 return f"doc_{document_id}_metadata"
235def get_metadata_cache(document_id: int) -> MetadataCacheData | None:
236 """
237 Returns the cached document metadata for the given document ID, as long as the metadata
238 was cached once and the checksums have not changed
239 """
240 doc_key = get_metadata_cache_key(document_id)
241 doc_metadata: MetadataCacheData | None = cache.get(doc_key)
242 # The metadata exists in the cache
243 if doc_metadata is not None:
244 try:
245 doc = Document.objects.only(
246 "pk",
247 "checksum",
248 "archive_checksum",
249 "archive_filename",
250 ).get(pk=document_id)
251 # The original checksums match
252 # If it has one, the archive checksums match
253 # Then, we can use the metadata
254 if (
255 doc_metadata.original_checksum == doc.checksum
256 and doc.has_archive_version
257 and doc_metadata.archive_checksum is not None
258 and doc_metadata.archive_checksum == doc.archive_checksum
259 ):
260 # Refresh cache
261 cache.touch(doc_key, CACHE_50_MINUTES)
262 return doc_metadata
263 else: # pragma: no cover
264 # Something didn't match, delete the key
265 cache.delete(doc_key)
266 except Document.DoesNotExist: # pragma: no cover
267 # Basically impossible, but the key existed, but the Document didn't
268 cache.delete(doc_key)
269 return None
272def set_metadata_cache(
273 document: Document,
274 original_metadata: list,
275 archive_metadata: list | None,
276 *,
277 timeout=CACHE_50_MINUTES,
278) -> None:
279 """
280 Sets the metadata into cache for the given Document
281 """
282 doc_key = get_metadata_cache_key(document.pk)
283 cache.set(
284 doc_key,
285 MetadataCacheData(
286 document.checksum,
287 original_metadata,
288 document.archive_checksum,
289 archive_metadata,
290 ),
291 timeout,
292 )
295def refresh_metadata_cache(
296 document_id: int,
297 *,
298 timeout: int = CACHE_50_MINUTES,
299) -> None:
300 """
301 Refreshes the expiration of the metadata for the given document ID
302 to the given timeout
303 """
304 doc_key = get_metadata_cache_key(document_id)
305 cache.touch(doc_key, timeout)
308def get_thumbnail_modified_key(document_id: int) -> str:
309 """
310 Builds the key to store a thumbnail's timestamp
311 """
312 return f"doc_{document_id}_thumbnail_modified"
315def clear_document_caches(document_id: int) -> None:
316 """
317 Removes all cached items for the given document
318 """
319 cache.delete_many(
320 [
321 get_suggestion_cache_key(document_id),
322 get_metadata_cache_key(document_id),
323 get_thumbnail_modified_key(document_id),
324 ],
325 )
326 invalidate_llm_suggestions_cache(document_id)