Coverage for api/utils/search_context.py: 94%

27 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 06:14 +0000

1from dataclasses import asdict, dataclass 

2from typing import Self 

3 

4from django.conf import settings 

5 

6from elasticsearch_dsl import Q, Search 

7 

8from api.constants.media_types import OriginIndex 

9from api.controllers.elasticsearch.helpers import get_es_response 

10 

11 

12@dataclass 

13class SearchContext: 

14 # Note: These sets use "identifiers" very explicitly 

15 # to convey that it is the Openverse result identifier and 

16 # not the document _id 

17 

18 all_result_identifiers: list[str] 

19 """All the result identifiers gathered for the search.""" 

20 

21 sensitive_text_result_identifiers: set[str] 

22 """Subset of result identifiers for results with sensitive textual content.""" 

23 

24 @classmethod 

25 def build( 

26 cls, all_result_identifiers: list[str], origin_index: OriginIndex 

27 ) -> Self: 

28 if not all_result_identifiers: 

29 return cls(list(), set()) 

30 

31 if not settings.ENABLE_FILTERED_INDEX_QUERIES: 31 ↛ 32line 31 didn't jump to line 32 because the condition on line 31 was never true

32 return cls(all_result_identifiers, set()) 

33 

34 filtered_index_search = Search(index=f"{origin_index}-filtered") 

35 filtered_index_search = filtered_index_search.query( 

36 # Use `identifier` rather than the document `id` due to 

37 # `id` instability between refreshes: 

38 # https://github.com/WordPress/openverse/issues/2306 

39 Q("terms", identifier=all_result_identifiers) 

40 ) 

41 

42 # The default query size is 10, so we need to slice the query 

43 # to change the size to be big enough to encompass all the 

44 # results. 

45 filtered_index_slice = filtered_index_search[: len(all_result_identifiers)] 

46 results_in_filtered_index = get_es_response( 

47 filtered_index_slice, es_query="filtered_index_context" 

48 ) 

49 filtered_index_identifiers = { 

50 result.identifier for result in results_in_filtered_index 

51 } 

52 sensitive_text_result_identifiers = { 

53 identifier 

54 for identifier in all_result_identifiers 

55 if identifier not in filtered_index_identifiers 

56 } 

57 

58 return cls( 

59 all_result_identifiers=all_result_identifiers, 

60 sensitive_text_result_identifiers=sensitive_text_result_identifiers, 

61 ) 

62 

63 def asdict(self): 

64 """ 

65 Cast the object to a dict. 

66 

67 This is a convenience method to avoid leaking dataclass 

68 implementation details elsewhere in the code. 

69 """ 

70 return asdict(self)