Coverage for documents/search/_schema.py: 56%
99 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import hashlib
4import json
5import logging
6import shutil
7from typing import TYPE_CHECKING
8from typing import Final
9from typing import NamedTuple
10from typing import cast
12import tantivy
13from django.conf import settings
14from whoosh_compat import FieldKind
16from documents.search._fields import PUBLIC_FIELDS
18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true
19 from pathlib import Path
21logger = logging.getLogger("paperless.search")
23# v1 - Initial tantivy schema format
24# v2 - build_schema() derived from PUBLIC_FIELDS, changing the field declaration
25# order, and the write-only correspondent/document_type/storage_path/tag id
26# columns dropped. tantivy compares schemas by ordered field list, so an
27# index built by v1 rejects every write against the v2 schema.
28# v3 - barcodes JSON field for stored barcode contents
29SCHEMA_VERSION: Final[int] = 3
32class FieldDescriptor(NamedTuple):
33 """One tantivy field, in declaration order.
35 The descriptor vocabulary is paperless', not tantivy-py's: it is both the
36 input to the SchemaBuilder and the input to schema_fingerprint(), so the
37 persisted fingerprint cannot move under a tantivy-py upgrade.
38 """
40 name: str
41 kind: str
42 stored: bool
43 indexed: bool
44 fast: bool
45 tokenizer: str | None
48# (schema kind, tokenizer) for the FieldKind -> FieldDescriptor mapping that
49# doesn't need special-casing. JSON is handled separately below since it can
50# emit a second, synthetic descriptor.
51_KIND_TABLE: Final[dict[FieldKind, tuple[str, str | None]]] = {
52 FieldKind.TEXT: ("text", "paperless_text"),
53 FieldKind.KEYWORD: ("text", "raw"),
54 FieldKind.U64: ("u64", None),
55 FieldKind.DATE: ("date", None),
56 FieldKind.DATETIME: ("date", None),
57}
58# Kinds whose fast-field flag follows FieldSpec.fast rather than always False.
59_FAST_FROM_FIELD: Final[frozenset[FieldKind]] = frozenset(
60 {FieldKind.U64, FieldKind.DATE, FieldKind.DATETIME},
61)
64def _public_field_descriptors() -> list[FieldDescriptor]:
65 """Descriptors for the query-visible fields declared in PUBLIC_FIELDS."""
66 descriptors: list[FieldDescriptor] = []
67 for field in PUBLIC_FIELDS:
68 if field.kind is FieldKind.JSON:
69 descriptors.append(
70 FieldDescriptor(
71 field.name,
72 "json",
73 stored=True,
74 indexed=True,
75 fast=False,
76 tokenizer="paperless_text",
77 ),
78 )
79 if field.name == "notes":
80 # Plain-text companion for snippet generation: tantivy's
81 # SnippetGenerator does not support JSON fields. Schema-only,
82 # no query-syntax meaning, not in PUBLIC_FIELDS.
83 descriptors.append(
84 FieldDescriptor(
85 "notes_text",
86 "text",
87 stored=True,
88 indexed=True,
89 fast=False,
90 tokenizer="paperless_text",
91 ),
92 )
93 continue
94 schema_kind, tokenizer = _KIND_TABLE[field.kind]
95 descriptors.append(
96 FieldDescriptor(
97 field.name,
98 schema_kind,
99 stored=True,
100 indexed=True,
101 fast=field.fast if field.kind in _FAST_FROM_FIELD else False,
102 tokenizer=tokenizer,
103 ),
104 )
105 return descriptors
108def field_descriptors() -> list[FieldDescriptor]:
109 """Every field of the document index, in the order tantivy declares them.
111 tantivy compares schemas by *ordered* field list, so the order here is
112 part of the on-disk contract: schema_fingerprint() hashes it and
113 needs_rebuild() acts on the result.
114 """
115 return [
116 FieldDescriptor(
117 "id",
118 "u64",
119 stored=True,
120 indexed=True,
121 fast=True,
122 tokenizer=None,
123 ),
124 *_public_field_descriptors(),
125 # Shadow sort fields - fast, not stored
126 *(
127 FieldDescriptor(
128 name,
129 "text",
130 stored=False,
131 indexed=True,
132 fast=True,
133 tokenizer="simple_analyzer",
134 )
135 for name in ("title_sort", "correspondent_sort", "type_sort")
136 ),
137 # CJK support - not stored, indexed only
138 *(
139 FieldDescriptor(
140 name,
141 "text",
142 stored=False,
143 indexed=True,
144 fast=False,
145 tokenizer="bigram_analyzer",
146 )
147 for name in (
148 "bigram_content",
149 "bigram_title",
150 "bigram_correspondent",
151 "bigram_document_type",
152 "bigram_tag",
153 )
154 ),
155 # Simple substring search support for title/content - not stored,
156 # indexed only
157 *(
158 FieldDescriptor(
159 name,
160 "text",
161 stored=False,
162 indexed=True,
163 fast=False,
164 tokenizer="simple_search_analyzer",
165 )
166 for name in ("simple_title", "simple_content")
167 ),
168 # Autocomplete prefix scan via terms_with_prefix, which walks the
169 # field's term dictionary - so the field must be indexed (term dict),
170 # not stored. The stored value is never read back, so storing it only
171 # wastes space.
172 FieldDescriptor(
173 "autocomplete_word",
174 "text",
175 stored=False,
176 indexed=True,
177 fast=False,
178 tokenizer="raw",
179 ),
180 # Permission filter columns, read by build_permission_filter.
181 *(
182 FieldDescriptor(
183 name,
184 "u64",
185 stored=False,
186 indexed=True,
187 fast=True,
188 tokenizer=None,
189 )
190 for name in ("owner_id", "viewer_id", "viewer_group_id")
191 ),
192 ]
195def schema_fingerprint() -> str:
196 """Hash of the field descriptors, stamped into .index_settings.json.
198 Changes whenever a field is added, removed, retyped, re-optioned or
199 reordered, so an index built from a different schema shape is detected
200 even when SCHEMA_VERSION was not bumped.
201 """
202 payload = json.dumps([list(descriptor) for descriptor in field_descriptors()])
203 return hashlib.blake2b(payload.encode()).hexdigest()
206def build_schema() -> tantivy.Schema:
207 """
208 Build the Tantivy schema for the paperless document index.
210 Creates a comprehensive schema supporting full-text search, filtering,
211 sorting, and autocomplete functionality. Includes fields for document
212 content, metadata, permissions, custom fields, and notes.
214 Returns:
215 Configured Tantivy schema ready for index creation
216 """
217 sb = tantivy.SchemaBuilder()
219 for descriptor in field_descriptors():
220 if descriptor.kind == "text":
221 sb.add_text_field(
222 descriptor.name,
223 stored=descriptor.stored,
224 fast=descriptor.fast,
225 tokenizer_name=cast("str", descriptor.tokenizer),
226 )
227 elif descriptor.kind == "json":
228 sb.add_json_field(
229 descriptor.name,
230 stored=descriptor.stored,
231 fast=descriptor.fast,
232 tokenizer_name=cast("str", descriptor.tokenizer),
233 )
234 elif descriptor.kind == "u64":
235 sb.add_unsigned_field(
236 descriptor.name,
237 stored=descriptor.stored,
238 indexed=descriptor.indexed,
239 fast=descriptor.fast,
240 )
241 elif descriptor.kind == "date": 241 ↛ 249line 241 didn't jump to line 249 because the condition on line 241 was always true
242 sb.add_date_field(
243 descriptor.name,
244 stored=descriptor.stored,
245 indexed=descriptor.indexed,
246 fast=descriptor.fast,
247 )
248 else:
249 raise ValueError(f"Unknown schema field kind: {descriptor.kind}")
251 return sb.build()
254def needs_rebuild(index_dir: Path) -> bool:
255 """
256 Check if the search index needs rebuilding.
258 Reads .index_settings.json to compare the stored schema version, search
259 language and schema fingerprint against the current configuration. Returns
260 True if the file is missing, unparsable, or any value mismatches.
262 Args:
263 index_dir: Path to the search index directory
265 Returns:
266 True if the index needs rebuilding, False if it's up to date
267 """
268 settings_file = index_dir / ".index_settings.json"
269 if not settings_file.exists():
270 return True
271 try:
272 data = json.loads(settings_file.read_text())
273 if data.get("schema_version") != SCHEMA_VERSION:
274 logger.info("Search index schema version mismatch - rebuilding.")
275 return True
276 if "language" not in data or data["language"] != settings.SEARCH_LANGUAGE:
277 logger.info("Search index language changed - rebuilding.")
278 return True
279 if data.get("schema_fingerprint") != schema_fingerprint():
280 logger.info("Search index schema fingerprint mismatch - rebuilding.")
281 return True
282 except ValueError:
283 return True
284 return False
287def wipe_index(index_dir: Path) -> None:
288 """
289 Delete all contents of the index directory to prepare for rebuild.
291 Recursively removes all files and subdirectories within the index
292 directory while preserving the directory itself.
294 Args:
295 index_dir: Path to the search index directory to clear
296 """
297 for child in index_dir.iterdir():
298 if child.is_dir():
299 shutil.rmtree(child)
300 else:
301 child.unlink()
304def _write_sentinels(index_dir: Path) -> None:
305 """Write .index_settings.json so the next index open can skip rebuilding."""
306 settings_file = index_dir / ".index_settings.json"
307 settings_file.write_text(
308 json.dumps(
309 {
310 "schema_version": SCHEMA_VERSION,
311 "language": settings.SEARCH_LANGUAGE,
312 "schema_fingerprint": schema_fingerprint(),
313 },
314 ),
315 )
318def open_or_rebuild_index(index_dir: Path | None = None) -> tantivy.Index:
319 """
320 Open the Tantivy index, creating or rebuilding as needed.
322 Checks if the index needs rebuilding due to schema version or language
323 changes. If rebuilding is needed, wipes the directory and creates a fresh
324 index with the current schema and configuration.
326 Args:
327 index_dir: Path to index directory (defaults to settings.INDEX_DIR)
329 Returns:
330 Opened Tantivy index (caller must register custom tokenizers)
331 """
332 if index_dir is None: 332 ↛ 333line 332 didn't jump to line 333 because the condition on line 332 was never true
333 index_dir = cast("Path", settings.INDEX_DIR)
334 if not index_dir.exists(): 334 ↛ 336line 334 didn't jump to line 336 because the condition on line 334 was always true
335 return tantivy.Index(build_schema())
336 if needs_rebuild(index_dir):
337 wipe_index(index_dir)
338 idx = tantivy.Index(build_schema(), path=str(index_dir))
339 _write_sentinels(index_dir)
340 return idx
341 try:
342 return tantivy.Index.open(str(index_dir))
343 except ValueError:
344 logger.exception(
345 "Search index is corrupted or incomplete - rebuilding from scratch.",
346 )
347 wipe_index(index_dir)
348 idx = tantivy.Index(build_schema(), path=str(index_dir))
349 _write_sentinels(index_dir)
350 return idx