Coverage for open_webui/retrieval/loaders/local.py: 28%

69 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1from importlib import import_module 

2from pathlib import Path 

3 

4from bs4 import BeautifulSoup 

5from langchain_core.documents import Document 

6 

7 

8class TextLoader: 

9 def __init__(self, file_path, encoding=None): 

10 self.file_path = str(file_path) 

11 self.encoding = encoding 

12 

13 def load(self) -> list[Document]: 

14 try: 

15 text = Path(self.file_path).read_text(encoding=self.encoding) 

16 except Exception as e: 

17 raise RuntimeError(f'Error loading {self.file_path}') from e 

18 return [ 

19 Document( 

20 page_content=text, 

21 metadata={'source': self.file_path}, 

22 ) 

23 ] 

24 

25 

26class HTMLLoader(TextLoader): 

27 def load(self) -> list[Document]: 

28 with open(self.file_path, encoding=self.encoding) as file: 

29 soup = BeautifulSoup(file, 'lxml') 

30 return [ 

31 Document( 

32 page_content=soup.get_text(), 

33 metadata={'source': self.file_path, 'title': str(soup.title.string) if soup.title else ''}, 

34 ) 

35 ] 

36 

37 

38class DocxLoader(TextLoader): 

39 def load(self) -> list[Document]: 

40 import docx2txt 

41 

42 return [ 

43 Document( 

44 page_content=docx2txt.process(Path(self.file_path).expanduser()), 

45 metadata={'source': self.file_path}, 

46 ) 

47 ] 

48 

49 

50class UnstructuredLoader: 

51 def __init__(self, file_path, file_format, mode='single', **kwargs): 

52 # Match the optional-package check; format dependencies are loaded when parsing. 

53 import_module('unstructured') 

54 self.file_path = file_path 

55 self.file_format = file_format 

56 self.mode = mode 

57 self.kwargs = kwargs 

58 

59 def load(self) -> list[Document]: 

60 file_format = self.file_format 

61 if file_format in ('doc', 'ppt', 'pptx'): 

62 from unstructured.file_utils.filetype import detect_filetype 

63 

64 legacy_format = 'doc' if file_format == 'doc' else 'ppt' 

65 try: 

66 import_module('magic') 

67 except ImportError: 

68 is_legacy = Path(self.file_path).suffix == f'.{legacy_format}' 

69 else: 

70 is_legacy = detect_filetype(self.file_path).name.lower() == legacy_format 

71 file_format = legacy_format if is_legacy else legacy_format + 'x' 

72 elif file_format == 'msg': 

73 from unstructured.file_utils.filetype import detect_filetype 

74 

75 detected = detect_filetype(self.file_path).name 

76 if detected not in ('EML', 'MSG'): 

77 raise ValueError(f'Unsupported email file type: {detected}') 

78 file_format = 'email' if detected == 'EML' else 'msg' 

79 

80 module = import_module(f'unstructured.partition.{file_format}') 

81 elements = getattr(module, f'partition_{file_format}')(filename=self.file_path, **self.kwargs) 

82 metadata = {'source': str(self.file_path)} 

83 if self.mode == 'elements': 

84 return [ 

85 Document( 

86 page_content=str(element), 

87 metadata={ 

88 **metadata, 

89 **element.metadata.to_dict(), 

90 'category': element.category, 

91 'element_id': element.id, 

92 }, 

93 ) 

94 for element in elements 

95 ] 

96 return [Document(page_content='\n\n'.join(map(str, elements)), metadata=metadata)] 

97 

98 

99class DocumentIntelligenceLoader: 

100 def __init__(self, file_path, api_endpoint, api_key=None, azure_credential=None, api_model='prebuilt-layout'): 

101 if (api_key is None) == (azure_credential is None): 

102 raise ValueError('Provide exactly one of api_key or azure_credential.') 

103 self.file_path = file_path 

104 self.api_endpoint = api_endpoint 

105 self.api_key = api_key 

106 self.azure_credential = azure_credential 

107 self.api_model = api_model 

108 

109 def load(self) -> list[Document]: 

110 from azure.ai.documentintelligence import DocumentIntelligenceClient 

111 from azure.core.credentials import AzureKeyCredential 

112 

113 credential = self.azure_credential if self.azure_credential is not None else AzureKeyCredential(self.api_key) 

114 with DocumentIntelligenceClient(self.api_endpoint, credential) as client, open(self.file_path, 'rb') as file: 

115 result = client.begin_analyze_document( 

116 self.api_model, 

117 body=file, 

118 content_type='application/octet-stream', 

119 output_content_format='markdown', 

120 ).result() 

121 return [Document(page_content=result.content, metadata=result.as_dict())]