Coverage for documents/search/_registry.py: 89%
31 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import dataclasses
4from typing import TYPE_CHECKING
6from whoosh_compat import FieldKind
7from whoosh_compat import FieldRegistry
9from documents.search._fields import PUBLIC_FIELDS
10from documents.search._tokenizer import ascii_fold
11from documents.search._tokenizer import paperless_text_analyzer
12from documents.search._tokenizer import stem_pattern_text
14if TYPE_CHECKING: 14 ↛ 15line 14 didn't jump to line 15 because the condition on line 14 was never true
15 from whoosh_compat import PatternNormalizer
17_registry_cache: dict[str | None, FieldRegistry] = {}
20def _identity_analyzer(text: str) -> list[str]:
21 """Analyzer for KEYWORD fields indexed with the raw tokenizer (no splitting)."""
22 return [text]
25def _fold_normalizer(text: str) -> str:
26 """Wildcard/regex literal-run normalizer for fields indexed without stemming."""
27 return ascii_fold(text.lower())
30def _make_pattern_normalizer(language: str | None) -> PatternNormalizer:
31 """Build the wildcard/regex literal-run normalizer for a search language."""
33 def _pattern_normalizer(text: str) -> tuple[str, ...]:
34 """Normalize a literal run into the forms a term may match.
36 TEXT index terms go through lowercase -> ascii_fold -> stem, so a
37 pattern that skips stemming can never match one: "invoice*" would look
38 for a term starting with "invoice" while the index holds "invoic". The
39 run is therefore offered stemmed as well. KEYWORD fields are indexed
40 raw and get _fold_normalizer instead, so their patterns stay literal.
42 Both forms are returned, as alternatives, because neither is a prefix
43 of the other in general: English stemming substitutes as well as
44 truncates ("copy" -> "copi"), so the stem alone loses the compounds
45 the typed run reaches ("copyright") while the typed run alone loses
46 the inflections the stem reaches ("copies"). whoosh-compat ORs the
47 alternatives per literal run and deduplicates them, so a run the
48 stemmer leaves alone costs exactly the one branch it did before.
50 Inside a bracket class the emitter calls this once per character and
51 uses the answer only if it is a single one-character form; two forms
52 there leave the character as typed. A stemmer does not change a lone
53 character, so the two forms deduplicate to one and the class body is
54 folded as before.
55 """
56 folded = ascii_fold(text.lower())
57 stemmed = stem_pattern_text(folded, language)
58 return (folded, stemmed)
60 return _pattern_normalizer
63def get_field_registry(language: str | None) -> FieldRegistry:
64 """Build (or return the cached) FieldRegistry for the given search language.
66 Cached keyed by language, rebuilt on the same trigger register_tokenizers()
67 uses (settings.SEARCH_LANGUAGE change). A fresh call with a new language
68 builds and caches a new registry rather than mutating the old one.
69 """
70 if language in _registry_cache:
71 return _registry_cache[language]
73 text_analyzer = paperless_text_analyzer(language).analyze
74 pattern_normalizer = _make_pattern_normalizer(language)
76 specs = [
77 dataclasses.replace(
78 field,
79 analyzer=_identity_analyzer
80 if field.kind is FieldKind.KEYWORD
81 else text_analyzer,
82 pattern_normalizer=_fold_normalizer
83 if field.kind is FieldKind.KEYWORD
84 else pattern_normalizer,
85 )
86 for field in PUBLIC_FIELDS
87 ]
89 registry = FieldRegistry(specs)
90 _registry_cache[language] = registry
91 return registry