Coverage for documents/search/_schema.py: 56%

99 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import hashlib 

4import json 

5import logging 

6import shutil 

7from typing import TYPE_CHECKING 

8from typing import Final 

9from typing import NamedTuple 

10from typing import cast 

11 

12import tantivy 

13from django.conf import settings 

14from whoosh_compat import FieldKind 

15 

16from documents.search._fields import PUBLIC_FIELDS 

17 

18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true

19 from pathlib import Path 

20 

21logger = logging.getLogger("paperless.search") 

22 

23# v1 - Initial tantivy schema format 

24# v2 - build_schema() derived from PUBLIC_FIELDS, changing the field declaration 

25# order, and the write-only correspondent/document_type/storage_path/tag id 

26# columns dropped. tantivy compares schemas by ordered field list, so an 

27# index built by v1 rejects every write against the v2 schema. 

28# v3 - barcodes JSON field for stored barcode contents 

29SCHEMA_VERSION: Final[int] = 3 

30 

31 

32class FieldDescriptor(NamedTuple): 

33 """One tantivy field, in declaration order. 

34 

35 The descriptor vocabulary is paperless', not tantivy-py's: it is both the 

36 input to the SchemaBuilder and the input to schema_fingerprint(), so the 

37 persisted fingerprint cannot move under a tantivy-py upgrade. 

38 """ 

39 

40 name: str 

41 kind: str 

42 stored: bool 

43 indexed: bool 

44 fast: bool 

45 tokenizer: str | None 

46 

47 

48# (schema kind, tokenizer) for the FieldKind -> FieldDescriptor mapping that 

49# doesn't need special-casing. JSON is handled separately below since it can 

50# emit a second, synthetic descriptor. 

51_KIND_TABLE: Final[dict[FieldKind, tuple[str, str | None]]] = { 

52 FieldKind.TEXT: ("text", "paperless_text"), 

53 FieldKind.KEYWORD: ("text", "raw"), 

54 FieldKind.U64: ("u64", None), 

55 FieldKind.DATE: ("date", None), 

56 FieldKind.DATETIME: ("date", None), 

57} 

58# Kinds whose fast-field flag follows FieldSpec.fast rather than always False. 

59_FAST_FROM_FIELD: Final[frozenset[FieldKind]] = frozenset( 

60 {FieldKind.U64, FieldKind.DATE, FieldKind.DATETIME}, 

61) 

62 

63 

64def _public_field_descriptors() -> list[FieldDescriptor]: 

65 """Descriptors for the query-visible fields declared in PUBLIC_FIELDS.""" 

66 descriptors: list[FieldDescriptor] = [] 

67 for field in PUBLIC_FIELDS: 

68 if field.kind is FieldKind.JSON: 

69 descriptors.append( 

70 FieldDescriptor( 

71 field.name, 

72 "json", 

73 stored=True, 

74 indexed=True, 

75 fast=False, 

76 tokenizer="paperless_text", 

77 ), 

78 ) 

79 if field.name == "notes": 

80 # Plain-text companion for snippet generation: tantivy's 

81 # SnippetGenerator does not support JSON fields. Schema-only, 

82 # no query-syntax meaning, not in PUBLIC_FIELDS. 

83 descriptors.append( 

84 FieldDescriptor( 

85 "notes_text", 

86 "text", 

87 stored=True, 

88 indexed=True, 

89 fast=False, 

90 tokenizer="paperless_text", 

91 ), 

92 ) 

93 continue 

94 schema_kind, tokenizer = _KIND_TABLE[field.kind] 

95 descriptors.append( 

96 FieldDescriptor( 

97 field.name, 

98 schema_kind, 

99 stored=True, 

100 indexed=True, 

101 fast=field.fast if field.kind in _FAST_FROM_FIELD else False, 

102 tokenizer=tokenizer, 

103 ), 

104 ) 

105 return descriptors 

106 

107 

108def field_descriptors() -> list[FieldDescriptor]: 

109 """Every field of the document index, in the order tantivy declares them. 

110 

111 tantivy compares schemas by *ordered* field list, so the order here is 

112 part of the on-disk contract: schema_fingerprint() hashes it and 

113 needs_rebuild() acts on the result. 

114 """ 

115 return [ 

116 FieldDescriptor( 

117 "id", 

118 "u64", 

119 stored=True, 

120 indexed=True, 

121 fast=True, 

122 tokenizer=None, 

123 ), 

124 *_public_field_descriptors(), 

125 # Shadow sort fields - fast, not stored 

126 *( 

127 FieldDescriptor( 

128 name, 

129 "text", 

130 stored=False, 

131 indexed=True, 

132 fast=True, 

133 tokenizer="simple_analyzer", 

134 ) 

135 for name in ("title_sort", "correspondent_sort", "type_sort") 

136 ), 

137 # CJK support - not stored, indexed only 

138 *( 

139 FieldDescriptor( 

140 name, 

141 "text", 

142 stored=False, 

143 indexed=True, 

144 fast=False, 

145 tokenizer="bigram_analyzer", 

146 ) 

147 for name in ( 

148 "bigram_content", 

149 "bigram_title", 

150 "bigram_correspondent", 

151 "bigram_document_type", 

152 "bigram_tag", 

153 ) 

154 ), 

155 # Simple substring search support for title/content - not stored, 

156 # indexed only 

157 *( 

158 FieldDescriptor( 

159 name, 

160 "text", 

161 stored=False, 

162 indexed=True, 

163 fast=False, 

164 tokenizer="simple_search_analyzer", 

165 ) 

166 for name in ("simple_title", "simple_content") 

167 ), 

168 # Autocomplete prefix scan via terms_with_prefix, which walks the 

169 # field's term dictionary - so the field must be indexed (term dict), 

170 # not stored. The stored value is never read back, so storing it only 

171 # wastes space. 

172 FieldDescriptor( 

173 "autocomplete_word", 

174 "text", 

175 stored=False, 

176 indexed=True, 

177 fast=False, 

178 tokenizer="raw", 

179 ), 

180 # Permission filter columns, read by build_permission_filter. 

181 *( 

182 FieldDescriptor( 

183 name, 

184 "u64", 

185 stored=False, 

186 indexed=True, 

187 fast=True, 

188 tokenizer=None, 

189 ) 

190 for name in ("owner_id", "viewer_id", "viewer_group_id") 

191 ), 

192 ] 

193 

194 

195def schema_fingerprint() -> str: 

196 """Hash of the field descriptors, stamped into .index_settings.json. 

197 

198 Changes whenever a field is added, removed, retyped, re-optioned or 

199 reordered, so an index built from a different schema shape is detected 

200 even when SCHEMA_VERSION was not bumped. 

201 """ 

202 payload = json.dumps([list(descriptor) for descriptor in field_descriptors()]) 

203 return hashlib.blake2b(payload.encode()).hexdigest() 

204 

205 

206def build_schema() -> tantivy.Schema: 

207 """ 

208 Build the Tantivy schema for the paperless document index. 

209 

210 Creates a comprehensive schema supporting full-text search, filtering, 

211 sorting, and autocomplete functionality. Includes fields for document 

212 content, metadata, permissions, custom fields, and notes. 

213 

214 Returns: 

215 Configured Tantivy schema ready for index creation 

216 """ 

217 sb = tantivy.SchemaBuilder() 

218 

219 for descriptor in field_descriptors(): 

220 if descriptor.kind == "text": 

221 sb.add_text_field( 

222 descriptor.name, 

223 stored=descriptor.stored, 

224 fast=descriptor.fast, 

225 tokenizer_name=cast("str", descriptor.tokenizer), 

226 ) 

227 elif descriptor.kind == "json": 

228 sb.add_json_field( 

229 descriptor.name, 

230 stored=descriptor.stored, 

231 fast=descriptor.fast, 

232 tokenizer_name=cast("str", descriptor.tokenizer), 

233 ) 

234 elif descriptor.kind == "u64": 

235 sb.add_unsigned_field( 

236 descriptor.name, 

237 stored=descriptor.stored, 

238 indexed=descriptor.indexed, 

239 fast=descriptor.fast, 

240 ) 

241 elif descriptor.kind == "date": 241 ↛ 249line 241 didn't jump to line 249 because the condition on line 241 was always true

242 sb.add_date_field( 

243 descriptor.name, 

244 stored=descriptor.stored, 

245 indexed=descriptor.indexed, 

246 fast=descriptor.fast, 

247 ) 

248 else: 

249 raise ValueError(f"Unknown schema field kind: {descriptor.kind}") 

250 

251 return sb.build() 

252 

253 

254def needs_rebuild(index_dir: Path) -> bool: 

255 """ 

256 Check if the search index needs rebuilding. 

257 

258 Reads .index_settings.json to compare the stored schema version, search 

259 language and schema fingerprint against the current configuration. Returns 

260 True if the file is missing, unparsable, or any value mismatches. 

261 

262 Args: 

263 index_dir: Path to the search index directory 

264 

265 Returns: 

266 True if the index needs rebuilding, False if it's up to date 

267 """ 

268 settings_file = index_dir / ".index_settings.json" 

269 if not settings_file.exists(): 

270 return True 

271 try: 

272 data = json.loads(settings_file.read_text()) 

273 if data.get("schema_version") != SCHEMA_VERSION: 

274 logger.info("Search index schema version mismatch - rebuilding.") 

275 return True 

276 if "language" not in data or data["language"] != settings.SEARCH_LANGUAGE: 

277 logger.info("Search index language changed - rebuilding.") 

278 return True 

279 if data.get("schema_fingerprint") != schema_fingerprint(): 

280 logger.info("Search index schema fingerprint mismatch - rebuilding.") 

281 return True 

282 except ValueError: 

283 return True 

284 return False 

285 

286 

287def wipe_index(index_dir: Path) -> None: 

288 """ 

289 Delete all contents of the index directory to prepare for rebuild. 

290 

291 Recursively removes all files and subdirectories within the index 

292 directory while preserving the directory itself. 

293 

294 Args: 

295 index_dir: Path to the search index directory to clear 

296 """ 

297 for child in index_dir.iterdir(): 

298 if child.is_dir(): 

299 shutil.rmtree(child) 

300 else: 

301 child.unlink() 

302 

303 

304def _write_sentinels(index_dir: Path) -> None: 

305 """Write .index_settings.json so the next index open can skip rebuilding.""" 

306 settings_file = index_dir / ".index_settings.json" 

307 settings_file.write_text( 

308 json.dumps( 

309 { 

310 "schema_version": SCHEMA_VERSION, 

311 "language": settings.SEARCH_LANGUAGE, 

312 "schema_fingerprint": schema_fingerprint(), 

313 }, 

314 ), 

315 ) 

316 

317 

318def open_or_rebuild_index(index_dir: Path | None = None) -> tantivy.Index: 

319 """ 

320 Open the Tantivy index, creating or rebuilding as needed. 

321 

322 Checks if the index needs rebuilding due to schema version or language 

323 changes. If rebuilding is needed, wipes the directory and creates a fresh 

324 index with the current schema and configuration. 

325 

326 Args: 

327 index_dir: Path to index directory (defaults to settings.INDEX_DIR) 

328 

329 Returns: 

330 Opened Tantivy index (caller must register custom tokenizers) 

331 """ 

332 if index_dir is None: 332 ↛ 333line 332 didn't jump to line 333 because the condition on line 332 was never true

333 index_dir = cast("Path", settings.INDEX_DIR) 

334 if not index_dir.exists(): 334 ↛ 336line 334 didn't jump to line 336 because the condition on line 334 was always true

335 return tantivy.Index(build_schema()) 

336 if needs_rebuild(index_dir): 

337 wipe_index(index_dir) 

338 idx = tantivy.Index(build_schema(), path=str(index_dir)) 

339 _write_sentinels(index_dir) 

340 return idx 

341 try: 

342 return tantivy.Index.open(str(index_dir)) 

343 except ValueError: 

344 logger.exception( 

345 "Search index is corrupted or incomplete - rebuilding from scratch.", 

346 ) 

347 wipe_index(index_dir) 

348 idx = tantivy.Index(build_schema(), path=str(index_dir)) 

349 _write_sentinels(index_dir) 

350 return idx