Coverage for documents/search/_query.py: 56%
233 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1from __future__ import annotations
3import functools
4import logging
5import unicodedata
6from functools import cache
7from typing import TYPE_CHECKING
8from typing import Final
10import regex
11import tantivy
12import whoosh_compat as wc
13from django.conf import settings
14from whoosh_compat.emitters.tantivy_ import emit as tantivy_emit
15from whoosh_compat.errors import Cause
16from whoosh_compat.errors import Diagnostic
17from whoosh_compat.errors import DiagnosticKind
18from whoosh_compat.errors import QueryError
20from documents.search._errors import InvalidDateQuery
21from documents.search._errors import InvalidNumberQuery
22from documents.search._errors import MultipleSearchQueryErrors
23from documents.search._errors import SearchQueryError
24from documents.search._registry import get_field_registry
25from documents.search._tokenizer import _bigram_analyzer
26from documents.search._tokenizer import paperless_text_analyzer
27from documents.search._tokenizer import simple_search_tokens
29if TYPE_CHECKING: 29 ↛ 30line 29 didn't jump to line 30 because the condition on line 29 was never true
30 from datetime import tzinfo
32logger = logging.getLogger("paperless.search")
34# Maximum seconds any single regex substitution over user-supplied query text
35# may run. The one remaining use is a character class, which cannot backtrack,
36# so the bound is an upper limit on that substitution's cost, not the ReDoS
37# guard it was originally written as.
38_REGEX_TIMEOUT: Final[float] = 1.0
40# Matches CJK/Hangul characters so queries can be routed to bigram fields.
41# Uses Unicode properties to cover all blocks including Extension B+ planes.
42# The marks that sit inside a Japanese word are listed explicitly, because
43# their Unicode script is Common and the script classes therefore miss
44# them: the katakana prolonged sound mark ー (U+30FC) and its halfwidth form
45# ー (U+FF70), the closing mark 〆 (U+3006), and the halfwidth voiced and
46# semi-voiced sound marks ゙ (U+FF9E) and ゚ (U+FF9F). Without them a word
47# splits into one-character runs, which have no bigrams: コーヒー becomes
48# コ + ヒ, and halfwidth パン becomes ハ + ン, leaving nothing to index or
49# search at all.
50#
51# The combining marks U+3099/U+309A are deliberately absent: everything
52# entering the index and every query string is put through
53# normalize_search_text first, so decomposed kana is composed away before
54# this pattern ever sees it. The halfwidth marks are not, and cannot be:
55# unlike パ (U+30D1), halfwidth katakana has no precomposed voiced form, so
56# NFC leaves ハ + ゚ as two codepoints where it folds か + U+3099 into が.
57_CJK_RE: Final = regex.compile(
58 r"[\p{Han}\p{Hiragana}\p{Katakana}\p{Hangul}ーー〆゙゚]+",
59)
62def normalize_search_text(text: str) -> str:
63 """Put text into the one Unicode normal form the index is built in.
65 Both the indexed text and the query string go through this, because a
66 bigram is a pair of codepoints: NFD がっこう is four where NFC is three,
67 so an unnormalized document never matches a normalized query.
69 NFC, not NFKC: folding パン to パン would be a search-behavior decision
70 rather than an encoding one.
71 """
72 return unicodedata.normalize("NFC", text)
75def _user_facing_emit_message(d: Diagnostic) -> str:
76 """A user-safe message for an emit-time QueryError's Diagnostic.
78 Built from the Diagnostic's structured fields (kind, field), never from
79 d.message: whoosh-compat documents that as developer/log output with no
80 stability guarantee, and PATTERN_TOO_COMPLEX embeds the raw backend
81 error text in it. SCHEMA_FIELD_MISSING never reaches here: _map_emit_error
82 re-raises it before calling this function, the same as INTERNAL.
83 """
84 field = str(d.field) if d.field is not None else None
85 if d.kind is DiagnosticKind.EXISTS_REQUIRES_FAST:
86 return f"Existence searches (field:*) are not supported for field {field!r}."
87 if d.kind is DiagnosticKind.TEXT_RANGE:
88 return f"Range searches are not supported for field {field!r}."
89 if d.kind is DiagnosticKind.PATTERN_TOO_COMPLEX:
90 return f"The wildcard pattern for field {field!r} is too complex."
91 logger.warning(
92 "Unmapped emit diagnostic %s: %s",
93 d.kind,
94 d.message,
95 ) # pragma: no cover
96 return "The search query could not be executed." # pragma: no cover
99def _map_emit_error(e: QueryError) -> SearchQueryError:
100 """Route an emit-time QueryError by its Diagnostic's Cause.
102 INVALID_INPUT/UNSUPPORTED are user-input errors, exactly like a parse
103 diagnostic, and map to a 400. INTERNAL means a defect in whoosh-compat
104 or in our own AST handling, never the user's query, so the QueryError is
105 re-raised rather than converted, reaching the generic 500 handler instead
106 of blaming the query. MISCONFIGURED other than EXISTS_REQUIRES_FAST is
107 treated the same way as INTERNAL: the registry and the index schema
108 disagree, which only an operator can fix, and the exact same query would
109 succeed on its own once the index is rebuilt. That makes it a transient
110 server-side condition, not a permanently bad request, so it is logged as
111 an error and re-raised rather than converted to a 400: telling the
112 client their query is invalid would be wrong, it would work once the
113 index catches up, and a 400 also hides the condition from monitoring
114 that only watches 5xx rates.
116 EXISTS_REQUIRES_FAST is the one MISCONFIGURED kind that is not a
117 disagreement. whoosh-compat derives it from the registry's own FieldSpec
118 (kind plus fast) without ever consulting the index schema, so it fires
119 whenever a non-fast field of a kind that cannot answer "exists" is asked
120 to: for us that is only the JSON fields, which field_descriptors() builds
121 non-fast on purpose. "notes:*" and the five other spellings of it are
122 ordinary user error that no operator action can clear, so they get the
123 400 without the alert.
124 """
125 d = e.diagnostic
126 if d.cause is Cause.INTERNAL:
127 raise e
128 if (
129 d.cause is Cause.MISCONFIGURED
130 and d.kind is not DiagnosticKind.EXISTS_REQUIRES_FAST
131 ):
132 logger.error(
133 "Search index misconfiguration for field %s (%s): %s",
134 d.field,
135 d.kind.name,
136 d.message,
137 )
138 raise e
139 return SearchQueryError(_user_facing_emit_message(d))
142def _has_cjk(text: str) -> bool:
143 """Return True if text contains any CJK characters."""
144 return bool(_CJK_RE.search(text))
147def extract_cjk_text(text: str) -> str:
148 """Join the CJK runs in ``text`` for indexing into bigram (char-ngram) fields.
150 Mirrors the query side, which extracts the CJK runs of whatever it is
151 about to search for (the raw string in simple modes, each CJK term's
152 own text in query mode): only CJK runs are ever searched against
153 the bigram fields, so only CJK runs are worth indexing there. Latin text
154 fed to a character-bigram field is never matched and only bloats the
155 index and slows indexing/merge. Returns "" when there is no CJK text.
156 """
157 return " ".join(_CJK_RE.findall(text))
160def _parse_cjk_text(
161 index: tantivy.Index,
162 cjk_text: str,
163 fields: list[str],
164) -> tantivy.Query | None:
165 """Parse a plain CJK run string against ``fields``, or None if it won't parse."""
166 try:
167 return index.parse_query(cjk_text, fields)
168 except Exception:
169 # Broad on purpose: cjk_text isn't filtered to a guaranteed-safe
170 # token set, so the exact failure mode tantivy could raise here
171 # isn't pinned down.
172 logger.debug(
173 "Skipping CJK search clause: could not parse CJK text: %r",
174 cjk_text,
175 )
176 return None
179def _build_cjk_query(
180 index: tantivy.Index,
181 raw_query: str,
182 fields: list[str],
183) -> tantivy.Query | None:
184 """Build a bigram-field query from the CJK runs in ``raw_query``.
186 For the simple (TEXT/TITLE) modes, whose input is plain text and carries
187 no query grammar to respect. Only the CJK character runs are extracted, so
188 a stray ``field:`` prefix or ``-``/``+`` in the input can neither leak
189 field semantics nor fail the parse, and no Latin token reaches the
190 character-bigram matcher (where it would produce spurious matches against
191 unrelated Latin text). Returns None when there is no CJK text or the parse
192 fails.
193 """
194 cjk_text = extract_cjk_text(raw_query)
195 if not cjk_text: 195 ↛ 196line 195 didn't jump to line 196 because the condition on line 195 was never true
196 return None
197 return _parse_cjk_text(index, cjk_text, fields)
200_DEFAULT_SEARCH_FIELDS: Final[list[str]] = [
201 "title",
202 "content",
203 "correspondent",
204 "document_type",
205 "tag",
206]
207_SIMPLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title", "simple_content"]
208_TITLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title"]
209# The bigram (character-ngram) companion of each default search field.
210_CJK_BIGRAM_FIELDS: Final[dict[str, str]] = {
211 field: f"bigram_{field}" for field in _DEFAULT_SEARCH_FIELDS
212}
213_CJK_CONTENT_FIELDS: Final[list[str]] = ["bigram_content"]
214_CJK_TITLE_FIELDS: Final[list[str]] = ["bigram_title"]
215_FIELD_BOOSTS = {"title": 2.0}
216_SIMPLE_FIELD_BOOSTS = {"simple_title": 2.0}
219@cache
220def _get_emit_field_registry(language: str | None) -> wc.FieldRegistry:
221 """The parse registry plus the CJK bigram fields, for analyzing and
222 emitting a query whose leaves _widen_leaf has widened. Cached per
223 language, on the same trigger get_field_registry() rebuilds on.
225 Never used to parse: the bigram fields are internal (absent from
226 PUBLIC_FIELDS), and queries are still parsed against
227 get_field_registry(), so ``bigram_content:...`` never becomes query
228 syntax.
230 The bigram specs set ``multitoken=Multitoken.AND`` explicitly. A
231 widened leaf's bigram side always sits inside the widening Or, so under
232 Multitoken.DEFAULT a run's bigrams would inherit that Or, and 東京都
233 would match a document containing only 京都.
234 """
235 bigram_analyze = _bigram_analyzer().analyze
236 return wc.FieldRegistry(
237 [
238 *get_field_registry(language),
239 *(
240 wc.FieldSpec(
241 bigram_field,
242 wc.FieldKind.TEXT,
243 analyzer=bigram_analyze,
244 multitoken=wc.Multitoken.AND,
245 )
246 for bigram_field in _CJK_BIGRAM_FIELDS.values()
247 ),
248 ],
249 )
252# Splits text exactly where the content analyzer's simple tokenizer does,
253# with none of its filters, so each piece is the raw text of one token the
254# index could hold. No remove_long either: a long CJK run must still reach
255# the bigram side.
256_TOKEN_SPLITTER: Final = tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple()).build()
259def _collapse(
260 node_cls: type[wc.ast.Node],
261 children: list[wc.ast.Node],
262 span: dict[str, int | None],
263) -> wc.ast.Node:
264 """Return the single child as it is, or wrap several in node_cls."""
265 if len(children) == 1:
266 return children[0]
267 return node_cls(children=tuple(children), **span)
270def _leaf_span(leaf: wc.ast.Term | wc.ast.Phrase) -> dict[str, int | None]:
271 """The startchar/endchar kwargs a leaf's alternatives are built with,
272 so a rewritten leaf still points at the same span of the original
273 query text."""
274 return {"startchar": leaf.startchar, "endchar": leaf.endchar}
277def _cjk_alternative(leaf: wc.ast.Term | wc.ast.Phrase) -> wc.ast.Node | None:
278 """Build the bigram alternative for a CJK leaf, or None if it gets none.
280 The content analyzer keeps an unspaced CJK run as one token, so only
281 the bigram fields can find a CJK term inside running text.
283 The alternative is built from the leaf's own text, split where the
284 content analyzer would split it. What each kind of piece contributes,
285 and how the pieces combine, is commented at the step that decides it.
286 None means there was nothing to build one from.
287 """
288 if leaf.field is None or leaf.field.name not in _CJK_BIGRAM_FIELDS: 288 ↛ 289line 288 didn't jump to line 289 because the condition on line 288 was never true
289 return None
290 text = str(leaf.text)
291 if not _has_cjk(text): 291 ↛ 292line 291 didn't jump to line 292 because the condition on line 291 was never true
292 return None
293 span = _leaf_span(leaf)
294 bigram_field = wc.FieldRef(_CJK_BIGRAM_FIELDS[leaf.field.name])
295 cjk_terms: list[wc.ast.Node] = []
296 latin_terms: list[wc.ast.Node] = []
297 for token in _TOKEN_SPLITTER.analyze(text):
298 runs = _CJK_RE.findall(token)
299 if runs:
300 # One bigram Term per run, never a joined string, which would
301 # produce bigrams spanning the join. A run's own bigrams stay
302 # jointly required through multitoken=AND on the bigram
303 # FieldSpec (see _get_emit_field_registry), which no enclosing
304 # group can loosen. A one-character run has no bigram at all
305 # and analyzes away to nothing.
306 cjk_terms.extend(
307 wc.ast.Term(field=bigram_field, text=run, **span) for run in runs
308 )
309 else:
310 # Latin the analyzer split off on its own. Latin glued to CJK
311 # inside one token (東京report) never reaches here, and must
312 # not: the index holds it only inside that whole unspaced
313 # token, so requiring it would lose documents the run finds.
314 latin_terms.append(wc.ast.Term(field=leaf.field, text=token, **span))
315 # A Term's runs are alternatives to each other, the way the separate
316 # bigram clause treated them. A Phrase's are required together: quoting
317 # asks for more than the bare words, and the parser's default group is
318 # And, so an Or here would make "東京都 大阪府" match strictly more
319 # than 東京都 大阪府 does. And is also the tightest thing available,
320 # since the bigram analyzer puts every token at position 0 and no
321 # alternative built from it can enforce adjacency.
322 cjk_group = wc.ast.And if isinstance(leaf, wc.ast.Phrase) else wc.ast.Or
323 pieces: list[wc.ast.Node] = []
324 if cjk_terms: 324 ↛ 326line 324 didn't jump to line 326 because the condition on line 324 was always true
325 pieces.append(_collapse(cjk_group, cjk_terms, span))
326 pieces.extend(latin_terms)
327 if not pieces: 327 ↛ 332line 327 didn't jump to line 332 because the condition on line 327 was never true
328 # A few hundred codepoints match _CJK_RE but yield no token at all
329 # from the simple tokenizer (CJK radicals, circled and squared
330 # forms), leaving nothing to widen with. The caller then leaves the
331 # leaf as it is, and it analyzes to the same nothing it does today.
332 return None
333 # Separated latin is required alongside the CJK side, which is what
334 # stops "invoice NOT 東京-report" from excluding every 東京 document.
335 return _collapse(wc.ast.And, pieces, span)
338# The index analyzer minus stemming: the same word boundaries and the same
339# drops (remove_long, characters the simple tokenizer discards). The field's
340# pattern_normalizer does the one stemming step.
341_FUZZY_WORD_SPLITTER: Final = paperless_text_analyzer(None)
344def _fuzzy_alternative(leaf: wc.ast.Term | wc.ast.Phrase) -> wc.ast.Node | None:
345 """Build the near-match alternative for a leaf, or None if it gets none.
347 Each of the leaf's words becomes a Fuzzy leaf on the leaf's own field.
348 A Term's words are OR'd, which is the per-word recall the old clause
349 had and what lets ``COVID-19`` match on one half. A Phrase's are
350 AND-ed: quoting asks for more than the bare words, so an Or there
351 would make a quoted phrase match strictly more than the same words
352 unquoted. Adjacency is out of reach either way, so requiring every
353 word is the floor.
355 Words come from the index analyzer minus its stemmer, so a leaf gets a
356 fuzzy side exactly when its exact side has tokens. A regex split would
357 keep words the index never holds (``__``, or a word past remove_long),
358 whose exact side analyzes to nothing, leaving a required fuzzy clause
359 that can never match.
361 Words of one character are skipped: with prefix matching, a
362 one-character fuzzy term matches every term in the field.
364 CJK words are skipped entirely. The content analyzer keeps an unspaced
365 CJK run as one token, so a prefix Fuzzy over it matches any run within
366 one edit of its start: ``東京`` would match a document holding only
367 ``京都の観光案内``, the very thing the bigram fields' multitoken=AND
368 exists to prevent (see _get_emit_field_registry). A two-character CJK
369 word is as broad here as the one-character word the length guard
370 already rejects, and _cjk_alternative supplies the in-run recall
371 anyway, so there is nothing to gain and precision to lose.
373 Fuzzy text goes through the field's pattern_normalizer rather than its
374 analyzer, and the splitter's output is already lowercased and folded,
375 so the word is stemmed exactly once.
376 """
377 words = [
378 word
379 for word in _FUZZY_WORD_SPLITTER.analyze(str(leaf.text))
380 if len(word) > 1 and not _has_cjk(word)
381 ]
382 if not words:
383 return None
384 span = _leaf_span(leaf)
385 group = wc.ast.And if isinstance(leaf, wc.ast.Phrase) else wc.ast.Or
386 leaves: list[wc.ast.Node] = [
387 wc.ast.Fuzzy(field=leaf.field, text=word, distance=1, prefix=True, **span)
388 for word in words
389 ]
390 return _collapse(group, leaves, span)
393def _widen_leaf(
394 leaf: wc.ast.Term | wc.ast.Phrase,
395 *,
396 fuzzy: bool,
397 negated: frozenset[int],
398) -> wc.ast.Node:
399 """``rewrite_leaf`` hook for emit(): widen a leaf on a default search
400 field to ``Or(leaf, alternatives...)`` where it sits, and leave every
401 other leaf as it is.
403 Widening in place, rather than OR-ing a separate clause in at the top,
404 keeps every AND, NOT, REQUIRE, boost, field restriction and positive
405 filter around the leaf applying to its widened match too.
407 A leaf can gain a CJK alternative, a fuzzy one, or both. A negated
408 leaf keeps its CJK alternative, because NOT X should exclude exactly
409 what X matches, but gets no fuzzy one: with prefix matching, NOT tax
410 would otherwise exclude "taxi" and "taxonomy". Negated leaves are
411 identified by identity through the pre-scan, since the hook cannot see
412 a leaf's context.
414 analyze() only offers Term and Phrase leaves, so Prefix and Wildcard
415 patterns are never widened.
416 """
417 if leaf.field is None or leaf.field.name not in _DEFAULT_SEARCH_FIELDS: 417 ↛ 418line 417 didn't jump to line 418 because the condition on line 417 was never true
418 return leaf
419 span = _leaf_span(leaf)
420 alternatives: list[wc.ast.Node] = []
421 cjk = _cjk_alternative(leaf)
422 if cjk is not None: 422 ↛ 424line 422 didn't jump to line 424 because the condition on line 422 was always true
423 alternatives.append(cjk)
424 if fuzzy and id(leaf) not in negated: 424 ↛ 425line 424 didn't jump to line 425 because the condition on line 424 was never true
425 near = _fuzzy_alternative(leaf)
426 if near is not None:
427 alternatives.append(wc.ast.Boosted(child=near, boost=0.1, **span))
428 if not alternatives: 428 ↛ 429line 428 didn't jump to line 429 because the condition on line 428 was never true
429 return leaf
430 # The leaf itself, not a copy: analyze() then keeps it combined the way
431 # its enclosing group says rather than the way this Or would, and a long
432 # run its analyzer drops to nothing leaves just the alternative.
433 return wc.ast.Or(children=(leaf, *alternatives), **span)
436def _negated_leaf_ids(node: wc.ast.Node) -> frozenset[int]:
437 """Return the id() of every Term and Phrase under a negation.
439 A negated leaf keeps its CJK alternative but gets no fuzzy one, and
440 the hook cannot see a leaf's context, so the tree is walked once here
441 and the hook compares by identity. analyze() guarantees it is handed
442 the input tree's own leaf objects, which is what makes identity work.
444 Negative positions are Not.child and AndNot.negative, and nothing
445 else in the node set. Leaves are collected at any depth and under any
446 number of negations: over-collecting costs a widening, while missing
447 a negated leaf would let NOT tax exclude "taxi".
449 Iterative, and total over node types: this runs outside emit()'s
450 error conversion, so an exception here would reach the generic 500
451 handler.
452 """
453 negated: set[int] = set()
454 stack: list[tuple[wc.ast.Node, bool]] = [(node, False)]
455 while stack:
456 current, under_negation = stack.pop()
457 if isinstance(current, (wc.ast.Term, wc.ast.Phrase)):
458 if under_negation:
459 negated.add(id(current))
460 elif isinstance(current, wc.ast.Not):
461 stack.append((current.child, True))
462 elif isinstance(current, wc.ast.AndNot):
463 stack.append((current.positive, under_negation))
464 stack.append((current.negative, True))
465 elif isinstance(current, wc.ast.AndMaybe):
466 stack.append((current.required, under_negation))
467 stack.append((current.optional, under_negation))
468 elif isinstance(current, wc.ast.Require):
469 stack.append((current.scored, under_negation))
470 stack.append((current.filter_only, under_negation))
471 elif isinstance(current, wc.ast.Boosted):
472 stack.append((current.child, under_negation))
473 elif isinstance(current, (wc.ast.And, wc.ast.Or)):
474 stack.extend((child, under_negation) for child in current.children)
475 return frozenset(negated)
478# Weight of the tiebreak clause that restores relevance ordering among
479# near-miss results. The widened tree is const-scored so the threshold
480# cannot cut a near-miss by the BM25 spread of the query's correctly
481# spelled words; this small share of its real score orders them again.
482# A BM25 spread above 0.1/_FUZZY_TIEBREAK can still cut one.
483_FUZZY_TIEBREAK: Final[float] = 0.01
486def _any_of(clauses: list[tuple[tantivy.Occur, tantivy.Query]]) -> tantivy.Query:
487 """Collapse a clause list: none -> empty, one -> itself (no wasted
488 single-clause boolean_query wrapping), many -> boolean_query(clauses)."""
489 if not clauses:
490 return tantivy.Query.empty_query()
491 if len(clauses) == 1:
492 return clauses[0][1]
493 return tantivy.Query.boolean_query(clauses)
496def _build_simple_token_query(
497 index: tantivy.Index,
498 fields: list[str],
499 token: str,
500 *,
501 allow_infix: bool,
502) -> tantivy.Query:
503 escaped = regex.escape(token)
504 # The simple analyzer keeps punctuation inside whitespace-delimited terms.
505 # Boundary-constrained query tokens may therefore begin either at the indexed
506 # term boundary or after punctuation within a term (for example,
507 # ``medical-history``). This avoids matching a numeric token such as ``6``
508 # in the middle of ``16``.
509 pattern = (
510 f".*{escaped}.*"
511 if allow_infix
512 else (
513 f"({escaped}.*|"
514 rf".*[\x20-\x2f\x3a-\x40\x5b-\x60\x7b-\x7e]{escaped}.*)"
515 )
516 )
517 field_queries: list[tuple[tantivy.Occur, tantivy.Query]] = []
518 for field in fields:
519 query = tantivy.Query.regex_query(index.schema, field, pattern)
520 boost = _SIMPLE_FIELD_BOOSTS.get(field, 1.0)
521 if boost > 1.0:
522 query = tantivy.Query.boost_query(query, boost)
523 field_queries.append((tantivy.Occur.Should, query))
525 return _any_of(field_queries)
528def parse_user_query(
529 index: tantivy.Index,
530 raw_query: str,
531 tz: tzinfo,
532) -> tantivy.Query:
533 """
534 Parse user query through whoosh-compat, then widen its leaves and emit,
535 once or twice depending on whether fuzzy matching is on.
537 1. wc.parse() against the shared FieldRegistry (whoosh grammar -> AST).
538 Bare notes:/custom_fields: prefixes resolve to their default subpath
539 (notes.note:/custom_fields.value:) directly in the registry, via
540 each JSON field's SubpathSpec(default=True).
541 2. Any diagnostics (bad dates/numbers) map to SearchQueryError subclasses
542 and raise, the view returns HTTP 400 with every offending field
543 listed, not just the first.
544 3. With fuzzy off, the AST is emitted once. If the query has CJK text,
545 emit()'s rewrite_leaf hook (_widen_leaf) rewrites each CJK term in
546 the AST to also match its bigram field, in place, so the rest of
547 the query constrains the bigram match too. With fuzzy on, the AST
548 is emitted twice through the same hook: once with fuzzy=True for
549 the widened tree used as a filter, and once with fuzzy=False for a
550 normally scored tree, and the two are blended into a boolean query
551 (see the comment above the blend for why). Either way the tree is
552 analyzed and emitted against _get_emit_field_registry() whenever
553 CJK or fuzzy widening applies, which adds the bigram fields; the
554 query itself was parsed without them.
555 emit() turns the AST into a tantivy.Query directly (no string
556 round-trip). A QueryError is routed by its Diagnostic's Cause
557 (_map_emit_error): a construct that parses but can't execute against
558 tantivy (e.g. a text-field range) is a 400, a registry/schema
559 mismatch is logged and re-raised, and an INTERNAL defect is
560 re-raised.
561 """
562 registry = get_field_registry(settings.SEARCH_LANGUAGE)
563 result = wc.parse(
564 raw_query,
565 registry=registry,
566 default_fields=_DEFAULT_SEARCH_FIELDS,
567 field_boosts=_FIELD_BOOSTS,
568 tz=tz,
569 )
570 if result.diagnostics: 570 ↛ 571line 570 didn't jump to line 571 because the condition on line 570 was never true
571 raise _diagnostics_to_error(result.diagnostics)
573 fuzzy_on = settings.ADVANCED_FUZZY_SEARCH_THRESHOLD is not None
574 cjk = _has_cjk(raw_query)
575 emit_registry = (
576 _get_emit_field_registry(settings.SEARCH_LANGUAGE)
577 if cjk or fuzzy_on
578 else registry
579 )
580 negated = _negated_leaf_ids(result.ast) if fuzzy_on else frozenset()
582 def emit_widened(*, fuzzy: bool) -> tantivy.Query:
583 hook = (
584 functools.partial(_widen_leaf, fuzzy=fuzzy, negated=negated)
585 if cjk or fuzzy
586 else None
587 )
588 return tantivy_emit(
589 result.ast,
590 index=index,
591 registry=emit_registry,
592 rewrite_leaf=hook,
593 )
595 try:
596 if not fuzzy_on: 596 ↛ 598line 596 didn't jump to line 598 because the condition on line 596 was always true
597 return emit_widened(fuzzy=False)
598 widened = emit_widened(fuzzy=True)
599 scored = emit_widened(fuzzy=False)
600 except QueryError as e:
601 raise _map_emit_error(e) from e
603 # The widened tree supplies the matched set at a flat score, so a
604 # near-miss is not ranked by how well the query's correctly spelled
605 # words matched. The CJK-only tree adds real scoring back for what
606 # matched exactly, and the last clause reintroduces the widened tree's
607 # own scoring at a small weight so near-misses still rank among
608 # themselves. const_score_query discards every boost inside it, the
609 # title field's 2.0 included, so a title match earns its boost through
610 # the second and third clauses rather than the first.
611 return tantivy.Query.boolean_query(
612 [
613 (tantivy.Occur.Must, tantivy.Query.const_score_query(widened, 0.1)),
614 (tantivy.Occur.Should, scored),
615 (
616 tantivy.Occur.Should,
617 tantivy.Query.boost_query(widened, _FUZZY_TIEBREAK),
618 ),
619 ],
620 )
623# The three whoosh-compat kinds for a wildcard on a field that cannot
624# carry one. d.field_kind supplies the discriminator, so naming the field's
625# type needs no second trip through the registry.
626_PATTERN_ON_KINDS: Final = frozenset(
627 {
628 DiagnosticKind.PATTERN_ON_NUMERIC,
629 DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS,
630 DiagnosticKind.PATTERN_ON_SUBPATH,
631 },
632)
635def _diagnostics_to_error(diagnostics: tuple[Diagnostic, ...]) -> SearchQueryError:
636 errors = [_single_diagnostic_to_error(d) for d in diagnostics]
637 return errors[0] if len(errors) == 1 else MultipleSearchQueryErrors(errors)
640def _single_diagnostic_to_error(d: Diagnostic) -> SearchQueryError:
641 # d.field is a FieldRef, not a str: str(d.field) gives the canonical
642 # dotted name (an aliased query, e.g. type:, reports document_type).
643 field_name = str(d.field) if d.field is not None else None
644 if d.kind is DiagnosticKind.BAD_DATE:
645 return InvalidDateQuery(field_name, d.raw_value)
646 if d.kind is DiagnosticKind.BAD_NUMBER:
647 return InvalidNumberQuery(field_name, d.raw_value)
648 if d.kind is DiagnosticKind.TOO_DEEP:
649 return SearchQueryError("The search query is nested too deeply.")
650 if d.kind in _PATTERN_ON_KINDS:
651 kind_label = f" ({d.field_kind.name.lower()})" if d.field_kind else ""
652 return SearchQueryError(
653 f"Wildcard patterns are not supported for field "
654 f"{field_name!r}{kind_label}.",
655 )
656 if d.kind is DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE:
657 field_label = f" for field {field_name!r}" if field_name else ""
658 return SearchQueryError(
659 f"{d.raw_value!r} looks like a bracket range{field_label}, but "
660 "'[' is not a wildcard character on its own. Combine it with a "
661 "wildcard, e.g. a trailing '*', or double-quote the value to "
662 "search it as literal text.",
663 )
664 logger.warning(
665 "Unmapped parse diagnostic %s: %s",
666 d.kind,
667 d.message,
668 ) # pragma: no cover
669 return SearchQueryError(
670 "The search query could not be executed.",
671 ) # pragma: no cover
674def parse_simple_query(
675 index: tantivy.Index,
676 raw_query: str,
677 fields: list[str],
678 cjk_fields: list[str] | None = None,
679) -> tantivy.Query:
680 """
681 Parse a plain-text query using Tantivy over a restricted field set.
683 Query string is escaped and normalized to be treated as "simple" text query.
684 When cjk_fields is provided and the query contains CJK characters, an
685 additional Should clause searches those bigram-tokenized fields, which match
686 CJK substrings the simple analyzer can't (long whitespace-free runs are
687 dropped by remove_long).
688 """
689 tokens = simple_search_tokens(raw_query)
691 clauses: list[tuple[tantivy.Occur, tantivy.Query]] = []
692 if tokens:
693 # Match every query token, regardless of its position in the document.
694 # Each token may occur in any of the requested fields, so text mode also
695 # finds documents whose matches are split between title and content.
696 token_queries = [
697 (
698 tantivy.Occur.Must,
699 _build_simple_token_query(
700 index,
701 fields,
702 token,
703 # Preserve historical infix matching for single-token
704 # searches. In multi-token searches, constrain numeric
705 # tokens to boundaries to avoid partial-number overlap.
706 # This depends on token content, not query order.
707 allow_infix=len(tokens) == 1 or not token.isdecimal(),
708 ),
709 )
710 for token in tokens
711 ]
712 clauses.append((tantivy.Occur.Should, _any_of(token_queries)))
714 if cjk_fields and _has_cjk(raw_query):
715 cjk_q = _build_cjk_query(index, raw_query, cjk_fields)
716 if cjk_q is not None:
717 clauses.append((tantivy.Occur.Should, cjk_q))
719 return _any_of(clauses)
722def parse_simple_text_highlight_query(
723 index: tantivy.Index,
724 raw_query: str,
725) -> tantivy.Query:
726 """Build a snippet-friendly query for simple text searches.
728 Simple search matching uses regex queries but for compatibility with Tantivy
729 SnippetGenerator we build a plain term query over the content field instead.
730 """
732 # Strip Tantivy operator chars before tokenizing: this is a plain-text
733 # highlight query, not a structured boolean query, so +/- are separators.
734 tokens = simple_search_tokens(
735 regex.sub(r"[-+]", " ", raw_query, timeout=_REGEX_TIMEOUT),
736 )
737 if not tokens:
738 return tantivy.Query.empty_query()
740 # Quote each token as its own phrase, escaping backslashes and embedded
741 # quotes. simple search tokens can carry arbitrary Tantivy syntax
742 # characters (`"`, `:`, `(`, `[`, `/`, ...) that the query-string parser
743 # would otherwise interpret as query grammar rather than literal text.
744 quoted_tokens = [
745 '"' + token.replace("\\", "\\\\").replace('"', '\\"') + '"' for token in tokens
746 ]
748 return index.parse_query(" ".join(quoted_tokens), ["content"])
751def parse_simple_text_query(
752 index: tantivy.Index,
753 raw_query: str,
754) -> tantivy.Query:
755 """
756 Parse a plain-text query over title/content for simple search inputs.
757 """
759 return parse_simple_query(
760 index,
761 raw_query,
762 _SIMPLE_SEARCH_FIELDS,
763 cjk_fields=_CJK_CONTENT_FIELDS,
764 )
767def parse_simple_title_query(
768 index: tantivy.Index,
769 raw_query: str,
770) -> tantivy.Query:
771 """
772 Parse a plain-text query over the title field only.
773 """
775 return parse_simple_query(
776 index,
777 raw_query,
778 _TITLE_SEARCH_FIELDS,
779 cjk_fields=_CJK_TITLE_FIELDS,
780 )