Coverage for open_webui/retrieval/loaders/local.py: 28%
69 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1from importlib import import_module
2from pathlib import Path
4from bs4 import BeautifulSoup
5from langchain_core.documents import Document
8class TextLoader:
9 def __init__(self, file_path, encoding=None):
10 self.file_path = str(file_path)
11 self.encoding = encoding
13 def load(self) -> list[Document]:
14 try:
15 text = Path(self.file_path).read_text(encoding=self.encoding)
16 except Exception as e:
17 raise RuntimeError(f'Error loading {self.file_path}') from e
18 return [
19 Document(
20 page_content=text,
21 metadata={'source': self.file_path},
22 )
23 ]
26class HTMLLoader(TextLoader):
27 def load(self) -> list[Document]:
28 with open(self.file_path, encoding=self.encoding) as file:
29 soup = BeautifulSoup(file, 'lxml')
30 return [
31 Document(
32 page_content=soup.get_text(),
33 metadata={'source': self.file_path, 'title': str(soup.title.string) if soup.title else ''},
34 )
35 ]
38class DocxLoader(TextLoader):
39 def load(self) -> list[Document]:
40 import docx2txt
42 return [
43 Document(
44 page_content=docx2txt.process(Path(self.file_path).expanduser()),
45 metadata={'source': self.file_path},
46 )
47 ]
50class UnstructuredLoader:
51 def __init__(self, file_path, file_format, mode='single', **kwargs):
52 # Match the optional-package check; format dependencies are loaded when parsing.
53 import_module('unstructured')
54 self.file_path = file_path
55 self.file_format = file_format
56 self.mode = mode
57 self.kwargs = kwargs
59 def load(self) -> list[Document]:
60 file_format = self.file_format
61 if file_format in ('doc', 'ppt', 'pptx'):
62 from unstructured.file_utils.filetype import detect_filetype
64 legacy_format = 'doc' if file_format == 'doc' else 'ppt'
65 try:
66 import_module('magic')
67 except ImportError:
68 is_legacy = Path(self.file_path).suffix == f'.{legacy_format}'
69 else:
70 is_legacy = detect_filetype(self.file_path).name.lower() == legacy_format
71 file_format = legacy_format if is_legacy else legacy_format + 'x'
72 elif file_format == 'msg':
73 from unstructured.file_utils.filetype import detect_filetype
75 detected = detect_filetype(self.file_path).name
76 if detected not in ('EML', 'MSG'):
77 raise ValueError(f'Unsupported email file type: {detected}')
78 file_format = 'email' if detected == 'EML' else 'msg'
80 module = import_module(f'unstructured.partition.{file_format}')
81 elements = getattr(module, f'partition_{file_format}')(filename=self.file_path, **self.kwargs)
82 metadata = {'source': str(self.file_path)}
83 if self.mode == 'elements':
84 return [
85 Document(
86 page_content=str(element),
87 metadata={
88 **metadata,
89 **element.metadata.to_dict(),
90 'category': element.category,
91 'element_id': element.id,
92 },
93 )
94 for element in elements
95 ]
96 return [Document(page_content='\n\n'.join(map(str, elements)), metadata=metadata)]
99class DocumentIntelligenceLoader:
100 def __init__(self, file_path, api_endpoint, api_key=None, azure_credential=None, api_model='prebuilt-layout'):
101 if (api_key is None) == (azure_credential is None):
102 raise ValueError('Provide exactly one of api_key or azure_credential.')
103 self.file_path = file_path
104 self.api_endpoint = api_endpoint
105 self.api_key = api_key
106 self.azure_credential = azure_credential
107 self.api_model = api_model
109 def load(self) -> list[Document]:
110 from azure.ai.documentintelligence import DocumentIntelligenceClient
111 from azure.core.credentials import AzureKeyCredential
113 credential = self.azure_credential if self.azure_credential is not None else AzureKeyCredential(self.api_key)
114 with DocumentIntelligenceClient(self.api_endpoint, credential) as client, open(self.file_path, 'rb') as file:
115 result = client.begin_analyze_document(
116 self.api_model,
117 body=file,
118 content_type='application/octet-stream',
119 output_content_format='markdown',
120 ).result()
121 return [Document(page_content=result.content, metadata=result.as_dict())]