Coverage for documents/search/_registry.py: 89%

31 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import dataclasses 

4from typing import TYPE_CHECKING 

5 

6from whoosh_compat import FieldKind 

7from whoosh_compat import FieldRegistry 

8 

9from documents.search._fields import PUBLIC_FIELDS 

10from documents.search._tokenizer import ascii_fold 

11from documents.search._tokenizer import paperless_text_analyzer 

12from documents.search._tokenizer import stem_pattern_text 

13 

14if TYPE_CHECKING: 14 ↛ 15line 14 didn't jump to line 15 because the condition on line 14 was never true

15 from whoosh_compat import PatternNormalizer 

16 

17_registry_cache: dict[str | None, FieldRegistry] = {} 

18 

19 

20def _identity_analyzer(text: str) -> list[str]: 

21 """Analyzer for KEYWORD fields indexed with the raw tokenizer (no splitting).""" 

22 return [text] 

23 

24 

25def _fold_normalizer(text: str) -> str: 

26 """Wildcard/regex literal-run normalizer for fields indexed without stemming.""" 

27 return ascii_fold(text.lower()) 

28 

29 

30def _make_pattern_normalizer(language: str | None) -> PatternNormalizer: 

31 """Build the wildcard/regex literal-run normalizer for a search language.""" 

32 

33 def _pattern_normalizer(text: str) -> tuple[str, ...]: 

34 """Normalize a literal run into the forms a term may match. 

35 

36 TEXT index terms go through lowercase -> ascii_fold -> stem, so a 

37 pattern that skips stemming can never match one: "invoice*" would look 

38 for a term starting with "invoice" while the index holds "invoic". The 

39 run is therefore offered stemmed as well. KEYWORD fields are indexed 

40 raw and get _fold_normalizer instead, so their patterns stay literal. 

41 

42 Both forms are returned, as alternatives, because neither is a prefix 

43 of the other in general: English stemming substitutes as well as 

44 truncates ("copy" -> "copi"), so the stem alone loses the compounds 

45 the typed run reaches ("copyright") while the typed run alone loses 

46 the inflections the stem reaches ("copies"). whoosh-compat ORs the 

47 alternatives per literal run and deduplicates them, so a run the 

48 stemmer leaves alone costs exactly the one branch it did before. 

49 

50 Inside a bracket class the emitter calls this once per character and 

51 uses the answer only if it is a single one-character form; two forms 

52 there leave the character as typed. A stemmer does not change a lone 

53 character, so the two forms deduplicate to one and the class body is 

54 folded as before. 

55 """ 

56 folded = ascii_fold(text.lower()) 

57 stemmed = stem_pattern_text(folded, language) 

58 return (folded, stemmed) 

59 

60 return _pattern_normalizer 

61 

62 

63def get_field_registry(language: str | None) -> FieldRegistry: 

64 """Build (or return the cached) FieldRegistry for the given search language. 

65 

66 Cached keyed by language, rebuilt on the same trigger register_tokenizers() 

67 uses (settings.SEARCH_LANGUAGE change). A fresh call with a new language 

68 builds and caches a new registry rather than mutating the old one. 

69 """ 

70 if language in _registry_cache: 

71 return _registry_cache[language] 

72 

73 text_analyzer = paperless_text_analyzer(language).analyze 

74 pattern_normalizer = _make_pattern_normalizer(language) 

75 

76 specs = [ 

77 dataclasses.replace( 

78 field, 

79 analyzer=_identity_analyzer 

80 if field.kind is FieldKind.KEYWORD 

81 else text_analyzer, 

82 pattern_normalizer=_fold_normalizer 

83 if field.kind is FieldKind.KEYWORD 

84 else pattern_normalizer, 

85 ) 

86 for field in PUBLIC_FIELDS 

87 ] 

88 

89 registry = FieldRegistry(specs) 

90 _registry_cache[language] = registry 

91 return registry