Coverage for documents/search/_tokenizer.py: 84%
55 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import logging
4from functools import cache
5from typing import Final
7import tantivy
9logger = logging.getLogger("paperless.search")
11# Mapping of ISO 639-1 codes (and common aliases) -> Tantivy Snowball name
12_LANGUAGE_MAP: dict[str, str] = {
13 "ar": "Arabic",
14 "arabic": "Arabic",
15 "da": "Danish",
16 "danish": "Danish",
17 "nl": "Dutch",
18 "dutch": "Dutch",
19 "en": "English",
20 "english": "English",
21 "fi": "Finnish",
22 "finnish": "Finnish",
23 "fr": "French",
24 "french": "French",
25 "de": "German",
26 "german": "German",
27 "el": "Greek",
28 "greek": "Greek",
29 "hu": "Hungarian",
30 "hungarian": "Hungarian",
31 "it": "Italian",
32 "italian": "Italian",
33 "no": "Norwegian",
34 "norwegian": "Norwegian",
35 "pt": "Portuguese",
36 "portuguese": "Portuguese",
37 "ro": "Romanian",
38 "romanian": "Romanian",
39 "ru": "Russian",
40 "russian": "Russian",
41 "es": "Spanish",
42 "spanish": "Spanish",
43 "sv": "Swedish",
44 "swedish": "Swedish",
45 "ta": "Tamil",
46 "tamil": "Tamil",
47 "tr": "Turkish",
48 "turkish": "Turkish",
49}
51SUPPORTED_LANGUAGES: frozenset[str] = frozenset(_LANGUAGE_MAP)
52# Document.title is max_length=128, so use 129 as the limit for
53# Tantivy's remove_long filter
54_TOKEN_REMOVE_LONG_LIMIT: Final[int] = 129
57def register_tokenizers(index: tantivy.Index, language: str | None) -> None:
58 """
59 Register all custom tokenizers required by the paperless schema.
61 Must be called on every Index instance since Tantivy requires tokenizer
62 re-registration after each index open/creation. Registers tokenizers for
63 full-text search, sorting, CJK language support, and fast-field indexing.
65 Args:
66 index: Tantivy index instance to register tokenizers on
67 language: ISO 639-1 language code for stemming (None to disable)
69 Note:
70 simple_analyzer is registered as both a text and fast-field tokenizer
71 since sort shadow fields (title_sort, correspondent_sort, type_sort)
72 use fast=True and Tantivy requires fast-field tokenizers to exist
73 even for documents that omit those fields.
74 """
75 index.register_tokenizer("paperless_text", paperless_text_analyzer(language))
76 index.register_tokenizer("simple_analyzer", _simple_analyzer())
77 index.register_tokenizer("bigram_analyzer", _bigram_analyzer())
78 index.register_tokenizer("simple_search_analyzer", _simple_search_analyzer())
79 # Fast-field tokenizer required for fast=True text fields in the schema
80 index.register_fast_field_tokenizer("simple_analyzer", _simple_analyzer())
83def paperless_text_analyzer(language: str | None) -> tantivy.TextAnalyzer:
84 """Main full-text tokenizer for content, title, etc: simple -> remove_long(129) -> lowercase -> ascii_fold [-> stemmer]"""
85 builder = (
86 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple())
87 .filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT))
88 .filter(tantivy.Filter.lowercase())
89 .filter(tantivy.Filter.ascii_fold())
90 )
91 if language:
92 tantivy_lang = _LANGUAGE_MAP.get(language.lower())
93 if tantivy_lang: 93 ↛ 96line 93 didn't jump to line 96 because the condition on line 93 was always true
94 builder = builder.filter(tantivy.Filter.stemmer(tantivy_lang))
95 else:
96 logger.warning(
97 "Unsupported search language '%s' - stemming disabled. Supported: %s",
98 language,
99 ", ".join(sorted(SUPPORTED_LANGUAGES)),
100 )
101 return builder.build()
104@cache
105def _pattern_stemmer(language: str | None) -> tantivy.TextAnalyzer | None:
106 """The stemming tail of paperless_text_analyzer, over a whole literal run.
108 Same language gate and same Snowball stemmer paperless_text_analyzer
109 applies at index time, so query patterns follow SEARCH_LANGUAGE. Returns
110 None when that gate disables stemming; paperless_text_analyzer already
111 warns about an unsupported language, so this stays quiet.
113 The raw tokenizer keeps the run whole (a wildcard literal is a fragment,
114 not necessarily a word), and remove_long is kept so an over-long run is
115 treated the same way the index treats it.
116 """
117 if not language: 117 ↛ 118line 117 didn't jump to line 118 because the condition on line 117 was never true
118 return None
119 tantivy_lang = _LANGUAGE_MAP.get(language.lower())
120 if tantivy_lang is None: 120 ↛ 121line 120 didn't jump to line 121 because the condition on line 120 was never true
121 return None
122 return (
123 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.raw())
124 .filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT))
125 .filter(tantivy.Filter.stemmer(tantivy_lang))
126 .build()
127 )
130def stem_pattern_text(text: str, language: str | None) -> str:
131 """Stem an already lowercased/ascii-folded run the way index terms are.
133 Returns text unchanged when stemming is disabled for language, and also
134 when the stem step does not yield exactly one token: remove_long drops a run
135 past the length limit, leaving no stem to substitute. Falling back to the
136 text as typed is the safe direction for a pattern prefix, since it can only
137 be as narrow as it was before stemming was considered.
139 The raw tokenizer emits one token whatever the input and the stemmer is
140 1-to-1, so only the zero-token case can fire today; the guard covers both
141 counts so a tokenizer change cannot turn this into an IndexError.
142 """
143 analyzer = _pattern_stemmer(language)
144 if analyzer is None: 144 ↛ 145line 144 didn't jump to line 145 because the condition on line 144 was never true
145 return text
146 tokens = analyzer.analyze(text)
147 if len(tokens) != 1: 147 ↛ 148line 147 didn't jump to line 148 because the condition on line 147 was never true
148 return text
149 return tokens[0]
152def _simple_analyzer() -> tantivy.TextAnalyzer:
153 """Tokenizer for shadow sort fields (title_sort, correspondent_sort, type_sort): simple -> lowercase -> ascii_fold."""
154 return (
155 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple())
156 .filter(tantivy.Filter.lowercase())
157 .filter(tantivy.Filter.ascii_fold())
158 .build()
159 )
162def _bigram_analyzer() -> tantivy.TextAnalyzer:
163 """Enables substring search in CJK text: ngram(2,2) -> lowercase. CJK / no-whitespace language support."""
164 return (
165 tantivy.TextAnalyzerBuilder(
166 tantivy.Tokenizer.ngram(min_gram=2, max_gram=2, prefix_only=False),
167 )
168 .filter(tantivy.Filter.lowercase())
169 .build()
170 )
173def _simple_search_analyzer() -> tantivy.TextAnalyzer:
174 """Tokenizer for simple substring search fields: non-whitespace chunks -> remove_long(129) -> lowercase -> ascii_fold."""
175 return (
176 tantivy.TextAnalyzerBuilder(
177 tantivy.Tokenizer.regex(r"\S+"),
178 )
179 .filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT))
180 .filter(tantivy.Filter.lowercase())
181 .filter(tantivy.Filter.ascii_fold())
182 .build()
183 )
186# Shared analyzers for query-side normalization. They reuse the exact filters
187# applied at index time so query terms fold identically (single source of truth
188# for ASCII folding, instead of a separate Python implementation). tantivy-py's
189# TextAnalyzer.analyze clones internally per call, so these are safe to share.
190_SIMPLE_SEARCH_ANALYZER: Final = _simple_search_analyzer()
191# raw tokenizer keeps the whole input as one token, so this folds an arbitrary
192# string to ASCII exactly like the content tokenizers (ß->ss, ø->o, æ->ae, ...)
193# without splitting it - used for autocomplete words and prefixes.
194_ASCII_FOLD_ANALYZER: Final = (
195 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.raw())
196 .filter(tantivy.Filter.ascii_fold())
197 .build()
198)
201def simple_search_tokens(text: str) -> list[str]:
202 """Tokenize a query string exactly as simple_title/simple_content are indexed."""
203 return _SIMPLE_SEARCH_ANALYZER.analyze(text)
206# Autocomplete word extraction: tokenize -> lowercase -> ascii_fold in a single
207# Rust pass. Uses the simple tokenizer so extracted words match how document
208# content is actually indexed (the content tokenizer _paperless_text also uses
209# simple()), replacing a Python regex scan plus per-token folding.
210_AUTOCOMPLETE_ANALYZER: Final = (
211 tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple())
212 .filter(tantivy.Filter.lowercase())
213 .filter(tantivy.Filter.ascii_fold())
214 .build()
215)
218def autocomplete_tokens(text: str) -> list[str]:
219 """Tokenize text into normalized autocomplete words (lowercased, ascii-folded)."""
220 return _AUTOCOMPLETE_ANALYZER.analyze(text)
223def ascii_fold(text: str) -> str:
224 """Fold text to ASCII using the same mapping as the content tokenizers.
226 Maps non-decomposable letters (ß->ss, ø->o, æ->ae, ...) identically to
227 Tantivy's ascii_fold filter used at index time, so query/autocomplete terms
228 agree with the folded content. A naive NFD strip would instead delete those
229 letters, causing silent search misses. Callers lowercase first, matching the
230 index pipeline's lowercase -> ascii_fold order.
231 """
232 tokens = _ASCII_FOLD_ANALYZER.analyze(text)
233 return tokens[0] if tokens else ""