Coverage for documents/search/_tokenizer.py: 84%

55 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1from __future__ import annotations 

2 

3import logging 

4from functools import cache 

5from typing import Final 

6 

7import tantivy 

8 

9logger = logging.getLogger("paperless.search") 

10 

11# Mapping of ISO 639-1 codes (and common aliases) -> Tantivy Snowball name 

12_LANGUAGE_MAP: dict[str, str] = { 

13 "ar": "Arabic", 

14 "arabic": "Arabic", 

15 "da": "Danish", 

16 "danish": "Danish", 

17 "nl": "Dutch", 

18 "dutch": "Dutch", 

19 "en": "English", 

20 "english": "English", 

21 "fi": "Finnish", 

22 "finnish": "Finnish", 

23 "fr": "French", 

24 "french": "French", 

25 "de": "German", 

26 "german": "German", 

27 "el": "Greek", 

28 "greek": "Greek", 

29 "hu": "Hungarian", 

30 "hungarian": "Hungarian", 

31 "it": "Italian", 

32 "italian": "Italian", 

33 "no": "Norwegian", 

34 "norwegian": "Norwegian", 

35 "pt": "Portuguese", 

36 "portuguese": "Portuguese", 

37 "ro": "Romanian", 

38 "romanian": "Romanian", 

39 "ru": "Russian", 

40 "russian": "Russian", 

41 "es": "Spanish", 

42 "spanish": "Spanish", 

43 "sv": "Swedish", 

44 "swedish": "Swedish", 

45 "ta": "Tamil", 

46 "tamil": "Tamil", 

47 "tr": "Turkish", 

48 "turkish": "Turkish", 

49} 

50 

51SUPPORTED_LANGUAGES: frozenset[str] = frozenset(_LANGUAGE_MAP) 

52# Document.title is max_length=128, so use 129 as the limit for 

53# Tantivy's remove_long filter 

54_TOKEN_REMOVE_LONG_LIMIT: Final[int] = 129 

55 

56 

57def register_tokenizers(index: tantivy.Index, language: str | None) -> None: 

58 """ 

59 Register all custom tokenizers required by the paperless schema. 

60 

61 Must be called on every Index instance since Tantivy requires tokenizer 

62 re-registration after each index open/creation. Registers tokenizers for 

63 full-text search, sorting, CJK language support, and fast-field indexing. 

64 

65 Args: 

66 index: Tantivy index instance to register tokenizers on 

67 language: ISO 639-1 language code for stemming (None to disable) 

68 

69 Note: 

70 simple_analyzer is registered as both a text and fast-field tokenizer 

71 since sort shadow fields (title_sort, correspondent_sort, type_sort) 

72 use fast=True and Tantivy requires fast-field tokenizers to exist 

73 even for documents that omit those fields. 

74 """ 

75 index.register_tokenizer("paperless_text", paperless_text_analyzer(language)) 

76 index.register_tokenizer("simple_analyzer", _simple_analyzer()) 

77 index.register_tokenizer("bigram_analyzer", _bigram_analyzer()) 

78 index.register_tokenizer("simple_search_analyzer", _simple_search_analyzer()) 

79 # Fast-field tokenizer required for fast=True text fields in the schema 

80 index.register_fast_field_tokenizer("simple_analyzer", _simple_analyzer()) 

81 

82 

83def paperless_text_analyzer(language: str | None) -> tantivy.TextAnalyzer: 

84 """Main full-text tokenizer for content, title, etc: simple -> remove_long(129) -> lowercase -> ascii_fold [-> stemmer]""" 

85 builder = ( 

86 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple()) 

87 .filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT)) 

88 .filter(tantivy.Filter.lowercase()) 

89 .filter(tantivy.Filter.ascii_fold()) 

90 ) 

91 if language: 

92 tantivy_lang = _LANGUAGE_MAP.get(language.lower()) 

93 if tantivy_lang: 93 ↛ 96line 93 didn't jump to line 96 because the condition on line 93 was always true

94 builder = builder.filter(tantivy.Filter.stemmer(tantivy_lang)) 

95 else: 

96 logger.warning( 

97 "Unsupported search language '%s' - stemming disabled. Supported: %s", 

98 language, 

99 ", ".join(sorted(SUPPORTED_LANGUAGES)), 

100 ) 

101 return builder.build() 

102 

103 

104@cache 

105def _pattern_stemmer(language: str | None) -> tantivy.TextAnalyzer | None: 

106 """The stemming tail of paperless_text_analyzer, over a whole literal run. 

107 

108 Same language gate and same Snowball stemmer paperless_text_analyzer 

109 applies at index time, so query patterns follow SEARCH_LANGUAGE. Returns 

110 None when that gate disables stemming; paperless_text_analyzer already 

111 warns about an unsupported language, so this stays quiet. 

112 

113 The raw tokenizer keeps the run whole (a wildcard literal is a fragment, 

114 not necessarily a word), and remove_long is kept so an over-long run is 

115 treated the same way the index treats it. 

116 """ 

117 if not language: 117 ↛ 118line 117 didn't jump to line 118 because the condition on line 117 was never true

118 return None 

119 tantivy_lang = _LANGUAGE_MAP.get(language.lower()) 

120 if tantivy_lang is None: 120 ↛ 121line 120 didn't jump to line 121 because the condition on line 120 was never true

121 return None 

122 return ( 

123 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.raw()) 

124 .filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT)) 

125 .filter(tantivy.Filter.stemmer(tantivy_lang)) 

126 .build() 

127 ) 

128 

129 

130def stem_pattern_text(text: str, language: str | None) -> str: 

131 """Stem an already lowercased/ascii-folded run the way index terms are. 

132 

133 Returns text unchanged when stemming is disabled for language, and also 

134 when the stem step does not yield exactly one token: remove_long drops a run 

135 past the length limit, leaving no stem to substitute. Falling back to the 

136 text as typed is the safe direction for a pattern prefix, since it can only 

137 be as narrow as it was before stemming was considered. 

138 

139 The raw tokenizer emits one token whatever the input and the stemmer is 

140 1-to-1, so only the zero-token case can fire today; the guard covers both 

141 counts so a tokenizer change cannot turn this into an IndexError. 

142 """ 

143 analyzer = _pattern_stemmer(language) 

144 if analyzer is None: 144 ↛ 145line 144 didn't jump to line 145 because the condition on line 144 was never true

145 return text 

146 tokens = analyzer.analyze(text) 

147 if len(tokens) != 1: 147 ↛ 148line 147 didn't jump to line 148 because the condition on line 147 was never true

148 return text 

149 return tokens[0] 

150 

151 

152def _simple_analyzer() -> tantivy.TextAnalyzer: 

153 """Tokenizer for shadow sort fields (title_sort, correspondent_sort, type_sort): simple -> lowercase -> ascii_fold.""" 

154 return ( 

155 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple()) 

156 .filter(tantivy.Filter.lowercase()) 

157 .filter(tantivy.Filter.ascii_fold()) 

158 .build() 

159 ) 

160 

161 

162def _bigram_analyzer() -> tantivy.TextAnalyzer: 

163 """Enables substring search in CJK text: ngram(2,2) -> lowercase. CJK / no-whitespace language support.""" 

164 return ( 

165 tantivy.TextAnalyzerBuilder( 

166 tantivy.Tokenizer.ngram(min_gram=2, max_gram=2, prefix_only=False), 

167 ) 

168 .filter(tantivy.Filter.lowercase()) 

169 .build() 

170 ) 

171 

172 

173def _simple_search_analyzer() -> tantivy.TextAnalyzer: 

174 """Tokenizer for simple substring search fields: non-whitespace chunks -> remove_long(129) -> lowercase -> ascii_fold.""" 

175 return ( 

176 tantivy.TextAnalyzerBuilder( 

177 tantivy.Tokenizer.regex(r"\S+"), 

178 ) 

179 .filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT)) 

180 .filter(tantivy.Filter.lowercase()) 

181 .filter(tantivy.Filter.ascii_fold()) 

182 .build() 

183 ) 

184 

185 

186# Shared analyzers for query-side normalization. They reuse the exact filters 

187# applied at index time so query terms fold identically (single source of truth 

188# for ASCII folding, instead of a separate Python implementation). tantivy-py's 

189# TextAnalyzer.analyze clones internally per call, so these are safe to share. 

190_SIMPLE_SEARCH_ANALYZER: Final = _simple_search_analyzer() 

191# raw tokenizer keeps the whole input as one token, so this folds an arbitrary 

192# string to ASCII exactly like the content tokenizers (ß->ss, ø->o, æ->ae, ...) 

193# without splitting it - used for autocomplete words and prefixes. 

194_ASCII_FOLD_ANALYZER: Final = ( 

195 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.raw()) 

196 .filter(tantivy.Filter.ascii_fold()) 

197 .build() 

198) 

199 

200 

201def simple_search_tokens(text: str) -> list[str]: 

202 """Tokenize a query string exactly as simple_title/simple_content are indexed.""" 

203 return _SIMPLE_SEARCH_ANALYZER.analyze(text) 

204 

205 

206# Autocomplete word extraction: tokenize -> lowercase -> ascii_fold in a single 

207# Rust pass. Uses the simple tokenizer so extracted words match how document 

208# content is actually indexed (the content tokenizer _paperless_text also uses 

209# simple()), replacing a Python regex scan plus per-token folding. 

210_AUTOCOMPLETE_ANALYZER: Final = ( 

211 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple()) 

212 .filter(tantivy.Filter.lowercase()) 

213 .filter(tantivy.Filter.ascii_fold()) 

214 .build() 

215) 

216 

217 

218def autocomplete_tokens(text: str) -> list[str]: 

219 """Tokenize text into normalized autocomplete words (lowercased, ascii-folded).""" 

220 return _AUTOCOMPLETE_ANALYZER.analyze(text) 

221 

222 

223def ascii_fold(text: str) -> str: 

224 """Fold text to ASCII using the same mapping as the content tokenizers. 

225 

226 Maps non-decomposable letters (ß->ss, ø->o, æ->ae, ...) identically to 

227 Tantivy's ascii_fold filter used at index time, so query/autocomplete terms 

228 agree with the folded content. A naive NFD strip would instead delete those 

229 letters, causing silent search misses. Callers lowercase first, matching the 

230 index pipeline's lowercase -> ascii_fold order. 

231 """ 

232 tokens = _ASCII_FOLD_ANALYZER.analyze(text) 

233 return tokens[0] if tokens else ""