Coverage for open_webui/retrieval/loaders/main.py: 24%

366 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1import asyncio 

2import csv 

3import logging 

4import os 

5import sys 

6import zipfile 

7 

8import ftfy 

9import requests 

10from fastapi import HTTPException 

11from azure.identity import DefaultAzureCredential 

12from langchain_core.documents import Document 

13from open_webui.env import ( 

14 AIOHTTP_CLIENT_SESSION_SSL, 

15 GLOBAL_LOG_LEVEL, 

16 USE_SLIM, 

17 MINERU_MAX_MARKDOWN_BYTES, 

18 REQUESTS_VERIFY, 

19) 

20from open_webui.retrieval.loaders.datalab_marker import DatalabMarkerLoader 

21from open_webui.retrieval.loaders.external_document import ExternalDocumentLoader 

22from open_webui.retrieval.loaders.local import ( 

23 DocumentIntelligenceLoader, 

24 DocxLoader, 

25 HTMLLoader, 

26 TextLoader, 

27 UnstructuredLoader, 

28) 

29from open_webui.retrieval.loaders.mineru import MinerULoader 

30from open_webui.retrieval.loaders.mistral import MistralLoader 

31from open_webui.retrieval.loaders.paddleocr_vl import PADDLEOCR_VL_SUPPORTED_EXTENSIONS, PaddleOCRVLLoader 

32from open_webui.retrieval.loaders.pdf import PDFLoader 

33from open_webui.utils.headers import get_user_groups_for_custom_headers 

34from open_webui.utils.json_codec import JSONCodec 

35 

36logging.basicConfig(stream=sys.stdout, level=GLOBAL_LOG_LEVEL) 

37log = logging.getLogger(__name__) 

38 

39known_source_ext = [ 

40 'go', 

41 'py', 

42 'java', 

43 'sh', 

44 'bat', 

45 'ps1', 

46 'cmd', 

47 'js', 

48 'ts', 

49 'css', 

50 'cpp', 

51 'hpp', 

52 'h', 

53 'c', 

54 'cs', 

55 'ino', 

56 'sql', 

57 'log', 

58 'ini', 

59 'pl', 

60 'pm', 

61 'r', 

62 'dart', 

63 'dockerfile', 

64 'env', 

65 'php', 

66 'hs', 

67 'hsc', 

68 'lua', 

69 'nginxconf', 

70 'conf', 

71 'm', 

72 'mm', 

73 'plsql', 

74 'perl', 

75 'rb', 

76 'rs', 

77 'db2', 

78 'scala', 

79 'bash', 

80 'swift', 

81 'vue', 

82 'svelte', 

83 'ex', 

84 'exs', 

85 'erl', 

86 'tsx', 

87 'jsx', 

88 'hs', 

89 'lhs', 

90 'json', 

91 'yaml', 

92 'yml', 

93 'toml', 

94 'svg', 

95] 

96 

97known_archive_ext = {'docx', 'epub', 'odt', 'pptx', 'xlsx'} 

98known_archive_content_types = { 

99 'application/epub+zip', 

100 'application/vnd.oasis.opendocument.text', 

101 'application/vnd.openxmlformats-officedocument.presentationml.presentation', 

102 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', 

103 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', 

104} 

105 

106 

107class ExcelLoader: 

108 """Fallback Excel loader using pandas when unstructured is not installed.""" 

109 

110 def __init__(self, file_path): 

111 self.file_path = file_path 

112 

113 def load(self) -> list[Document]: 

114 import pandas as pd 

115 

116 text_parts = [] 

117 xls = pd.ExcelFile(self.file_path) 

118 for sheet_name in xls.sheet_names: 

119 df = pd.read_excel(xls, sheet_name=sheet_name) 

120 text_parts.append(f'Sheet: {sheet_name}\n{df.to_string(index=False)}') 

121 return [ 

122 Document( 

123 page_content='\n\n'.join(text_parts), 

124 metadata={'source': self.file_path}, 

125 ) 

126 ] 

127 

128 

129def get_csv_summary(filename: str, file_path: str, encoding: str) -> str | None: 

130 try: 

131 with open(file_path, newline='', encoding=encoding) as f: 

132 sample = f.read(4096) 

133 f.seek(0) 

134 try: 

135 dialect = csv.Sniffer().sniff(sample) 

136 except csv.Error: 

137 dialect = csv.excel 

138 

139 total_rows = 0 

140 max_columns = 0 

141 headers = [] 

142 for row in csv.reader(f, dialect): 

143 total_rows += 1 

144 max_columns = max(max_columns, len(row)) 

145 if total_rows == 1: 

146 headers = [header.lstrip('\ufeff') for header in row] 

147 except Exception: 

148 return None 

149 

150 if total_rows == 0: 

151 return None 

152 

153 return ( 

154 f'Table: {total_rows} rows incl. header; ' 

155 f'{max(total_rows - 1, 0)} data rows; ' 

156 f'{max_columns} columns: {", ".join(headers)}.' 

157 ) 

158 

159 

160class CSVLoaderWithSummary: 

161 def __init__(self, file_path: str, filename: str, encoding: str): 

162 self.file_path = file_path 

163 self.filename = filename 

164 self.encoding = encoding 

165 

166 def load(self) -> list[Document]: 

167 docs = [] 

168 try: 

169 with open(self.file_path, newline='', encoding=self.encoding) as file: 

170 for index, row in enumerate(csv.DictReader(file)): 

171 fields = [] 

172 for key, value in row.items(): 

173 if isinstance(value, str): 

174 value = value.strip() 

175 elif isinstance(value, list): 

176 value = ','.join(v.strip() for v in value) 

177 fields.append(f'{key.strip() if key is not None else key}: {value}') 

178 content = '\n'.join(fields) 

179 docs.append(Document(page_content=content, metadata={'source': self.file_path, 'row': index})) 

180 except Exception as e: 

181 raise RuntimeError(f'Error loading {self.file_path}') from e 

182 if os.getenv('ENABLE_RAG_CSV_SUMMARY', 'False').lower() == 'true': 

183 summary = get_csv_summary(self.filename, self.file_path, self.encoding) 

184 if summary: 

185 docs.insert(0, Document(page_content=summary, metadata={'source': self.file_path, 'row': -1})) 

186 return docs 

187 

188 

189class PptxLoader: 

190 """Fallback PowerPoint loader using python-pptx when unstructured is not installed.""" 

191 

192 def __init__(self, file_path): 

193 self.file_path = file_path 

194 

195 def load(self) -> list[Document]: 

196 from pptx import Presentation 

197 

198 prs = Presentation(self.file_path) 

199 text_parts = [] 

200 for i, slide in enumerate(prs.slides, 1): 

201 slide_texts = [] 

202 for shape in slide.shapes: 

203 if shape.has_text_frame: 

204 slide_texts.append(shape.text_frame.text) 

205 if slide_texts: 

206 text_parts.append(f'Slide {i}:\n' + '\n'.join(slide_texts)) 

207 return [ 

208 Document( 

209 page_content='\n\n'.join(text_parts), 

210 metadata={'source': self.file_path}, 

211 ) 

212 ] 

213 

214 

215class TikaLoader: 

216 def __init__(self, url, file_path, mime_type=None, extract_images=None, server_version='3'): 

217 self.url = url 

218 self.file_path = file_path 

219 self.mime_type = mime_type 

220 self.server_version = str(server_version or '3') 

221 

222 self.extract_images = extract_images 

223 

224 def load(self) -> list[Document]: 

225 with open(self.file_path, 'rb') as f: 

226 data = f.read() 

227 

228 if self.mime_type is not None: 

229 headers = {'Content-Type': self.mime_type} 

230 else: 

231 headers = {} 

232 

233 if self.extract_images == True: 

234 headers['X-Tika-PDFextractInlineImages'] = 'true' 

235 

236 endpoint_path = 'tika/json/md' if self.server_version == '4' else 'tika/text' 

237 content_key = 'tk:content' if self.server_version == '4' else 'X-TIKA:content' 

238 endpoint = f'{self.url.rstrip("/")}/{endpoint_path}' 

239 

240 r = requests.put(endpoint, data=data, headers=headers, verify=REQUESTS_VERIFY) 

241 

242 if r.ok: 

243 raw_metadata = r.json() 

244 text = raw_metadata.get(content_key, '<No text content found>').strip() 

245 

246 if 'Content-Type' in raw_metadata: 

247 headers['Content-Type'] = raw_metadata['Content-Type'] 

248 

249 log.debug('Tika extracted text: %s', text) 

250 

251 return [Document(page_content=text, metadata=headers)] 

252 else: 

253 raise Exception(f'Error calling Tika: {r.reason}') 

254 

255 

256class DoclingLoader: 

257 def __init__(self, url, api_key=None, file_path=None, mime_type=None, params=None): 

258 self.url = url.rstrip('/') 

259 self.api_key = api_key 

260 self.file_path = file_path 

261 self.mime_type = mime_type 

262 

263 self.params = params or {} 

264 

265 def load(self) -> list[Document]: 

266 page_break_marker = '\f' 

267 with open(self.file_path, 'rb') as f: 

268 headers = {} 

269 if self.api_key: 

270 headers['X-Api-Key'] = f'{self.api_key}' 

271 

272 r = requests.post( 

273 f'{self.url}/v1/convert/file', 

274 files={ 

275 'files': ( 

276 self.file_path, 

277 f, 

278 self.mime_type or 'application/octet-stream', 

279 ) 

280 }, 

281 data={ 

282 'image_export_mode': 'placeholder', 

283 'md_page_break_placeholder': page_break_marker, 

284 # Keep Docling params as user-provided form values. Encoding nested 

285 # values here would make Open WebUI responsible for Docling's API 

286 # quirks and could break when Docling changes its form contract. 

287 **self.params, 

288 }, 

289 headers=headers, 

290 verify=AIOHTTP_CLIENT_SESSION_SSL, 

291 ) 

292 if r.ok: 

293 result = r.json() 

294 # Docling reports failed and skipped conversions inside HTTP 200 responses. 

295 conversion_status = result.get('status') 

296 if conversion_status in ['failure', 'skipped']: 

297 error_details = ( 

298 '; '.join(filter(None, (error.get('error_message') for error in result.get('errors', [])))) 

299 or 'no error message provided' 

300 ) 

301 raise Exception(f'Error calling Docling: conversion status {conversion_status} - {error_details}') 

302 

303 document_data = result.get('document', {}) 

304 md_content = document_data.get('md_content') or '' 

305 text = md_content or '<No text content found>' 

306 

307 metadata = {'Content-Type': self.mime_type} if self.mime_type else {} 

308 if page_break_marker in md_content: 

309 documents = [ 

310 Document(page_content=page.strip(), metadata={**metadata, 'page': page_idx}) 

311 for page_idx, page in enumerate(md_content.split(page_break_marker)) 

312 if page.strip() 

313 ] 

314 if documents: 

315 log.debug('Docling extracted text: %s', text) 

316 return documents 

317 

318 log.debug('Docling extracted text: %s', text) 

319 return [Document(page_content=text, metadata=metadata)] 

320 else: 

321 error_msg = f'Error calling Docling API: {r.reason}' 

322 if r.text: 

323 try: 

324 error_data = r.json() 

325 if 'detail' in error_data: 

326 error_msg += f' - {error_data["detail"]}' 

327 except Exception: 

328 error_msg += f' - {r.text}' 

329 raise Exception(f'Error calling Docling: {error_msg}') 

330 

331 

332class Loader: 

333 def __init__(self, engine: str = '', **kwargs): 

334 self.engine = engine 

335 self.user = kwargs.get('user', None) 

336 self.user_groups = kwargs.get('user_groups', None) 

337 self.metadata = kwargs.get('metadata', {}) 

338 self.kwargs = kwargs 

339 

340 def load(self, filename: str, file_content_type: str, file_path: str) -> list[Document]: 

341 loader = self._get_loader(filename, file_content_type, file_path) 

342 docs = loader.load() 

343 # ftfy's auto mode unescapes entities on every line before the first literal '<', rewriting the document. 

344 return [ 

345 Document(page_content=ftfy.fix_text(doc.page_content, unescape_html=False), metadata=doc.metadata) 

346 for doc in docs 

347 ] 

348 

349 async def aload(self, filename: str, file_content_type: str, file_path: str) -> list[Document]: 

350 """ 

351 Async wrapper around `load`. 

352 

353 Document loaders dispatched by `_get_loader` (PyMuPDF, Unstructured, 

354 python-docx, Tika, etc.) are uniformly synchronous and CPU/IO-bound. 

355 Calling `load` directly from an async handler would block the event 

356 loop for the entire parse — minutes for large PDFs. This offloads 

357 the work to a worker thread so the loop stays responsive. 

358 """ 

359 # Group lookup is async-only, so it must happen before `load` 

360 # is offloaded to a thread without a running event loop. 

361 if self.engine == 'external' and self.user_groups is None: 361 ↛ 362line 361 didn't jump to line 362 because the condition on line 361 was never true

362 self.user_groups = await get_user_groups_for_custom_headers( 

363 self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_HEADERS'), self.user 

364 ) 

365 

366 return await asyncio.to_thread(self.load, filename, file_content_type, file_path) 

367 

368 def _is_text_file(self, file_ext: str, file_content_type: str) -> bool: 

369 return file_ext in known_source_ext or ( 

370 file_content_type 

371 and file_content_type.find('text/') >= 0 

372 # Avoid text/html files being detected as text 

373 and not file_content_type.find('html') >= 0 

374 ) 

375 

376 def _detect_text_encoding(self, file_path: str) -> str: 

377 """Detect the encoding of a text file with CJK-aware fallbacks. 

378 

379 Langchain's ``TextLoader`` uses chardet internally when 

380 ``autodetect_encoding=True``, but chardet frequently misidentifies 

381 CJK encodings (e.g. GB18030 detected as GB2312 or even Cyrillic). 

382 This method replaces that by: 

383 

384 1. Trying UTF-8 first (fast path for the vast majority of files). 

385 2. Using chardet as a *hint* to prioritise the right CJK codec 

386 family, but mapping subset names to their superset 

387 (e.g. GB2312 → gb18030). 

388 3. Validating that decoded text actually contains CJK characters, 

389 guarding against codecs that "succeed" but produce garbage. 

390 4. Falling back to latin-1 (always valid, ftfy fixes mojibake later). 

391 """ 

392 try: 

393 with open(file_path, 'rb') as f: 

394 raw = f.read() 

395 except OSError: 

396 return 'utf-8' 

397 

398 if not raw: 398 ↛ 399line 398 didn't jump to line 399 because the condition on line 398 was never true

399 return 'utf-8' 

400 

401 # Fast path: most files are UTF-8 

402 try: 

403 raw.decode('utf-8') 

404 return 'utf-8' 

405 except UnicodeDecodeError as e: 

406 first_non_utf8 = e.start 

407 

408 # Use chardet as a hint, not as ground truth 

409 import chardet 

410 

411 # chardet is pure Python (~1.3s/MB), so sample around the first bad byte 

412 window = 256 * 1024 

413 sample_start = max(0, first_non_utf8 - window // 2) 

414 sample = raw[sample_start : sample_start + window] 

415 detected = chardet.detect(sample) 

416 # A stray byte can sit far from the real payload, leaving the sample with nothing to read 

417 if len(sample.translate(None, delete=bytes(range(128)))) < 64 and len(sample) < len(raw): 

418 detected = chardet.detect(raw) 

419 detected_enc = (detected.get('encoding') or '').lower().replace('-', '').replace('_', '') 

420 

421 # Map chardet's detected encoding to the correct superset codec. 

422 # chardet often reports GB2312 for content that is actually GB18030; 

423 # GB18030 is a strict superset of both GB2312 and GBK. 

424 _ENC_FAMILY = { 

425 'gb2312': 'gb18030', 

426 'gb18030': 'gb18030', 

427 'gbk': 'gb18030', 

428 'big5': 'big5', 

429 'euckr': 'euc-kr', 

430 'eucjp': 'euc-jp', 

431 'iso2022jp': 'euc-jp', 

432 'shiftjis': 'shift_jis', 

433 } 

434 

435 # Build priority list: chardet-hinted codec first, then remaining CJK 

436 base_order = ['gb18030', 'big5', 'euc-kr', 'euc-jp'] 

437 hinted = _ENC_FAMILY.get(detected_enc) 

438 if hinted and hinted in base_order: 

439 ordered = [hinted] + [e for e in base_order if e != hinted] 

440 else: 

441 ordered = base_order 

442 

443 for enc in ordered: 

444 try: 

445 text = raw.decode(enc) 

446 if text.strip() and self._has_cjk_characters(text): 

447 log.info( 

448 'Detected encoding %s for %s (chardet guessed %s)', 

449 enc, 

450 file_path, 

451 detected.get('encoding'), 

452 ) 

453 return enc 

454 except (UnicodeDecodeError, LookupError): 

455 continue 

456 

457 # If chardet gave a non-CJK answer that isn't in our family map, 

458 # try it directly — it might be a valid Western encoding. 

459 chardet_encoding = detected.get('encoding') 

460 if chardet_encoding: 

461 try: 

462 raw.decode(chardet_encoding) 

463 log.info( 

464 'Using chardet-detected encoding %s for %s', 

465 chardet_encoding, 

466 file_path, 

467 ) 

468 return chardet_encoding 

469 except (UnicodeDecodeError, LookupError): 

470 pass 

471 

472 # latin-1 is the ultimate fallback: every byte 0x00–0xFF is valid. 

473 # ftfy.fix_text() (applied downstream) repairs most mojibake that 

474 # results from treating Windows-1252 content as Latin-1. 

475 log.info('Falling back to latin-1 encoding for %s', file_path) 

476 return 'latin-1' 

477 

478 @staticmethod 

479 def _has_cjk_characters(text: str, threshold: float = 0.05) -> bool: 

480 """Check if decoded text contains a meaningful proportion of CJK characters. 

481 

482 This guards against codecs that technically "succeed" but decode the 

483 bytes into wrong Unicode codepoints (e.g. PUA chars, random symbols). 

484 A genuine CJK document should have at least ``threshold`` fraction of 

485 its non-whitespace characters in CJK Unicode blocks. 

486 """ 

487 if not text: 

488 return False 

489 

490 cjk_count = 0 

491 total = 0 

492 for ch in text: 

493 if ch.isspace(): 

494 continue 

495 total += 1 

496 cp = ord(ch) 

497 if ( 

498 0x4E00 <= cp <= 0x9FFF # CJK Unified Ideographs 

499 or 0x3400 <= cp <= 0x4DBF # CJK Extension A 

500 or 0x20000 <= cp <= 0x2A6DF # CJK Extension B 

501 or 0x2A700 <= cp <= 0x2B73F # CJK Extension C 

502 or 0x2B740 <= cp <= 0x2B81F # CJK Extension D 

503 or 0xF900 <= cp <= 0xFAFF # CJK Compatibility Ideographs 

504 or 0x3000 <= cp <= 0x303F # CJK Symbols and Punctuation 

505 or 0x3040 <= cp <= 0x309F # Hiragana 

506 or 0x30A0 <= cp <= 0x30FF # Katakana 

507 or 0xAC00 <= cp <= 0xD7AF # Hangul Syllables 

508 or 0xFF00 <= cp <= 0xFFEF # Halfwidth and Fullwidth Forms 

509 ): 

510 cjk_count += 1 

511 

512 if total == 0: 

513 return False 

514 

515 return (cjk_count / total) >= threshold 

516 

517 def _get_loader(self, filename: str, file_content_type: str, file_path: str): 

518 file_ext = filename.split('.')[-1].lower() 

519 

520 if file_ext in known_archive_ext or file_content_type in known_archive_content_types: 520 ↛ 521line 520 didn't jump to line 521 because the condition on line 520 was never true

521 max_file_size = self.kwargs.get('FILE_MAX_SIZE') 

522 try: 

523 max_file_size_bytes = int(max_file_size) * 1024 * 1024 if max_file_size else 100 * 1024 * 1024 

524 except (TypeError, ValueError): 

525 max_file_size_bytes = 100 * 1024 * 1024 

526 

527 if max_file_size_bytes > 0: 

528 try: 

529 with zipfile.ZipFile(file_path) as archive: 

530 uncompressed_size = sum(entry.file_size for entry in archive.infolist()) 

531 except (zipfile.BadZipFile, OSError): 

532 pass 

533 else: 

534 max_bytes = min( 

535 max(10 * 1024 * 1024, os.path.getsize(file_path) * 100), 

536 max_file_size_bytes, 

537 ) 

538 if uncompressed_size > max_bytes: 

539 raise ValueError('Document archive is too large after decompression') 

540 

541 if ( 541 ↛ 546line 541 didn't jump to line 546 because the condition on line 541 was never true

542 self.engine == 'external' 

543 and self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_URL') 

544 and self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_API_KEY') 

545 ): 

546 loader = ExternalDocumentLoader( 

547 file_path=file_path, 

548 url=self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_URL'), 

549 api_key=self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_API_KEY'), 

550 mime_type=file_content_type, 

551 user=self.user, 

552 user_groups=self.user_groups, 

553 headers=self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_HEADERS'), 

554 metadata={ 

555 **self.metadata, 

556 'file_name': filename, 

557 'file_content_type': file_content_type, 

558 }, 

559 ) 

560 elif self.engine == 'tika' and self.kwargs.get('TIKA_SERVER_URL'): 560 ↛ 561line 560 didn't jump to line 561 because the condition on line 560 was never true

561 if self._is_text_file(file_ext, file_content_type): 

562 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

563 else: 

564 loader = TikaLoader( 

565 url=self.kwargs.get('TIKA_SERVER_URL'), 

566 file_path=file_path, 

567 server_version=self.kwargs.get('TIKA_SERVER_VERSION'), 

568 extract_images=self.kwargs.get('PDF_EXTRACT_IMAGES'), 

569 ) 

570 elif ( 570 ↛ 595line 570 didn't jump to line 595 because the condition on line 570 was never true

571 self.engine == 'datalab_marker' 

572 and self.kwargs.get('DATALAB_MARKER_API_KEY') 

573 and file_ext 

574 in [ 

575 'pdf', 

576 'xls', 

577 'xlsx', 

578 'ods', 

579 'doc', 

580 'docx', 

581 'odt', 

582 'ppt', 

583 'pptx', 

584 'odp', 

585 'html', 

586 'epub', 

587 'png', 

588 'jpeg', 

589 'jpg', 

590 'webp', 

591 'gif', 

592 'tiff', 

593 ] 

594 ): 

595 api_base_url = self.kwargs.get('DATALAB_MARKER_API_BASE_URL', '') 

596 if not api_base_url or api_base_url.strip() == '': 

597 api_base_url = 'https://www.datalab.to/api/v1/marker' # https://github.com/open-webui/open-webui/pull/16867#issuecomment-3218424349 

598 

599 loader = DatalabMarkerLoader( 

600 file_path=file_path, 

601 api_key=self.kwargs['DATALAB_MARKER_API_KEY'], 

602 api_base_url=api_base_url, 

603 additional_config=self.kwargs.get('DATALAB_MARKER_ADDITIONAL_CONFIG'), 

604 use_llm=self.kwargs.get('DATALAB_MARKER_USE_LLM', False), 

605 skip_cache=self.kwargs.get('DATALAB_MARKER_SKIP_CACHE', False), 

606 force_ocr=self.kwargs.get('DATALAB_MARKER_FORCE_OCR', False), 

607 paginate=self.kwargs.get('DATALAB_MARKER_PAGINATE', False), 

608 strip_existing_ocr=self.kwargs.get('DATALAB_MARKER_STRIP_EXISTING_OCR', False), 

609 disable_image_extraction=self.kwargs.get('DATALAB_MARKER_DISABLE_IMAGE_EXTRACTION', False), 

610 format_lines=self.kwargs.get('DATALAB_MARKER_FORMAT_LINES', False), 

611 output_format=self.kwargs.get('DATALAB_MARKER_OUTPUT_FORMAT', 'markdown'), 

612 ) 

613 elif self.engine == 'docling' and self.kwargs.get('DOCLING_SERVER_URL'): 613 ↛ 614line 613 didn't jump to line 614 because the condition on line 613 was never true

614 if self._is_text_file(file_ext, file_content_type): 

615 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

616 else: 

617 # Build params for DoclingLoader 

618 params = self.kwargs.get('DOCLING_PARAMS', {}) 

619 if not isinstance(params, dict): 

620 try: 

621 params = JSONCodec.loads(params) 

622 except JSONCodec.JSONDecodeError: 

623 log.error('Invalid DOCLING_PARAMS format, expected JSON object') 

624 params = {} 

625 

626 loader = DoclingLoader( 

627 url=self.kwargs.get('DOCLING_SERVER_URL'), 

628 api_key=self.kwargs.get('DOCLING_API_KEY', None), 

629 file_path=file_path, 

630 mime_type=file_content_type, 

631 params=params, 

632 ) 

633 elif ( 633 ↛ 646line 633 didn't jump to line 646 because the condition on line 633 was never true

634 self.engine == 'document_intelligence' 

635 and self.kwargs.get('DOCUMENT_INTELLIGENCE_ENDPOINT') != '' 

636 and ( 

637 file_ext in ['pdf', 'docx', 'ppt', 'pptx'] 

638 or file_content_type 

639 in [ 

640 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', 

641 'application/vnd.ms-powerpoint', 

642 'application/vnd.openxmlformats-officedocument.presentationml.presentation', 

643 ] 

644 ) 

645 ): 

646 if self.kwargs.get('DOCUMENT_INTELLIGENCE_KEY') != '': 

647 loader = DocumentIntelligenceLoader( 

648 file_path=file_path, 

649 api_endpoint=self.kwargs.get('DOCUMENT_INTELLIGENCE_ENDPOINT'), 

650 api_key=self.kwargs.get('DOCUMENT_INTELLIGENCE_KEY'), 

651 api_model=self.kwargs.get('DOCUMENT_INTELLIGENCE_MODEL'), 

652 ) 

653 else: 

654 loader = DocumentIntelligenceLoader( 

655 file_path=file_path, 

656 api_endpoint=self.kwargs.get('DOCUMENT_INTELLIGENCE_ENDPOINT'), 

657 azure_credential=DefaultAzureCredential(), 

658 api_model=self.kwargs.get('DOCUMENT_INTELLIGENCE_MODEL'), 

659 ) 

660 elif self.engine == 'mineru' and file_ext in self.kwargs.get('MINERU_FILE_EXTENSIONS', ['pdf']): 660 ↛ 661line 660 didn't jump to line 661 because the condition on line 660 was never true

661 mineru_timeout = self.kwargs.get('MINERU_API_TIMEOUT', 300) 

662 if mineru_timeout: 

663 try: 

664 mineru_timeout = int(mineru_timeout) 

665 except ValueError: 

666 mineru_timeout = 300 

667 loader = MinerULoader( 

668 file_path=file_path, 

669 api_mode=self.kwargs.get('MINERU_API_MODE', 'local'), 

670 api_url=self.kwargs.get('MINERU_API_URL', 'http://localhost:8000'), 

671 api_key=self.kwargs.get('MINERU_API_KEY', ''), 

672 params=self.kwargs.get('MINERU_PARAMS', {}), 

673 timeout=mineru_timeout, 

674 max_markdown_bytes=MINERU_MAX_MARKDOWN_BYTES, 

675 ) 

676 elif ( 676 ↛ 681line 676 didn't jump to line 681 because the condition on line 676 was never true

677 self.engine == 'mistral_ocr' 

678 and self.kwargs.get('MISTRAL_OCR_API_KEY') != '' 

679 and file_ext in ['pdf'] # Mistral OCR currently only supports PDF and images 

680 ): 

681 loader = MistralLoader( 

682 base_url=self.kwargs.get('MISTRAL_OCR_API_BASE_URL'), 

683 api_key=self.kwargs.get('MISTRAL_OCR_API_KEY'), 

684 file_path=file_path, 

685 use_base64=self.kwargs.get('MISTRAL_OCR_USE_BASE64', False), 

686 user=self.user, 

687 ) 

688 elif ( 688 ↛ 694line 688 didn't jump to line 694 because the condition on line 688 was never true

689 self.engine == 'paddleocr_vl' 

690 and self.kwargs.get('PADDLEOCR_VL_BASE_URL') 

691 and self.kwargs.get('PADDLEOCR_VL_TOKEN') 

692 and file_ext in PADDLEOCR_VL_SUPPORTED_EXTENSIONS 

693 ): 

694 loader = PaddleOCRVLLoader( 

695 api_url=self.kwargs.get('PADDLEOCR_VL_BASE_URL'), 

696 token=self.kwargs.get('PADDLEOCR_VL_TOKEN'), 

697 file_path=file_path, 

698 ) 

699 else: 

700 if USE_SLIM: 700 ↛ 701line 700 didn't jump to line 701 because the condition on line 700 was never true

701 if file_ext == 'csv': 

702 return CSVLoaderWithSummary(file_path, filename, self._detect_text_encoding(file_path)) 

703 if file_ext in ['htm', 'html']: 

704 return HTMLLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

705 if file_ext in ['txt', 'md', 'markdown', 'rst', 'xml'] or self._is_text_file( 

706 file_ext, file_content_type 

707 ): 

708 return TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

709 raise HTTPException( 

710 503, 

711 'This file type requires an external document extractor in slim. Configure one that supports it.', 

712 ) 

713 if file_ext == 'pdf': 713 ↛ 714line 713 didn't jump to line 714 because the condition on line 713 was never true

714 loader = PDFLoader( 

715 file_path, 

716 extract_images=self.kwargs.get('PDF_EXTRACT_IMAGES'), 

717 mode=self.kwargs.get('PDF_LOADER_MODE', 'page'), 

718 ) 

719 elif file_ext == 'csv': 719 ↛ 720line 719 didn't jump to line 720 because the condition on line 719 was never true

720 loader = CSVLoaderWithSummary( 

721 file_path, 

722 filename, 

723 self._detect_text_encoding(file_path), 

724 ) 

725 elif file_ext == 'rst': 725 ↛ 726line 725 didn't jump to line 726 because the condition on line 725 was never true

726 try: 

727 loader = UnstructuredLoader(file_path, 'rst', mode='elements') 

728 except ImportError: 

729 log.warning( 

730 "The 'unstructured' package is not installed. " 

731 'Falling back to plain text loading for .rst file. ' 

732 'Install it with: pip install unstructured' 

733 ) 

734 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

735 elif file_ext == 'xml': 735 ↛ 736line 735 didn't jump to line 736 because the condition on line 735 was never true

736 try: 

737 loader = UnstructuredLoader(file_path, 'xml') 

738 except ImportError: 

739 log.warning( 

740 "The 'unstructured' package is not installed. " 

741 'Falling back to plain text loading for .xml file. ' 

742 'Install it with: pip install unstructured' 

743 ) 

744 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

745 elif file_ext in ['htm', 'html']: 745 ↛ 746line 745 didn't jump to line 746 because the condition on line 745 was never true

746 loader = HTMLLoader(file_path, encoding='unicode_escape') 

747 elif file_ext == 'md': 747 ↛ 748line 747 didn't jump to line 748 because the condition on line 747 was never true

748 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

749 elif file_content_type == 'application/epub+zip': 749 ↛ 750line 749 didn't jump to line 750 because the condition on line 749 was never true

750 try: 

751 loader = UnstructuredLoader(file_path, 'epub') 

752 except ImportError: 

753 raise ValueError( 

754 "Processing .epub files requires the 'unstructured' package. " 

755 'Install it with: pip install unstructured' 

756 ) 

757 elif ( 757 ↛ 761line 757 didn't jump to line 761 because the condition on line 757 was never true

758 file_content_type == 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' 

759 or file_ext == 'docx' 

760 ): 

761 loader = DocxLoader(file_path) 

762 elif file_ext == 'doc' or file_content_type == 'application/msword': 762 ↛ 763line 762 didn't jump to line 763 because the condition on line 762 was never true

763 try: 

764 loader = UnstructuredLoader(file_path, 'doc') 

765 except ImportError: 

766 raise ValueError( 

767 "Processing .doc files requires the 'unstructured' package. " 

768 'Install it with: pip install unstructured' 

769 ) 

770 elif file_content_type in [ 770 ↛ 774line 770 didn't jump to line 774 because the condition on line 770 was never true

771 'application/vnd.ms-excel', 

772 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', 

773 ] or file_ext in ['xls', 'xlsx']: 

774 try: 

775 loader = UnstructuredLoader(file_path, 'xlsx') 

776 except ImportError: 

777 log.warning( 

778 "The 'unstructured' package is not installed. " 

779 'Falling back to pandas for Excel file loading. ' 

780 'Install unstructured for better results: pip install unstructured' 

781 ) 

782 loader = ExcelLoader(file_path) 

783 elif file_content_type in [ 783 ↛ 787line 783 didn't jump to line 787 because the condition on line 783 was never true

784 'application/vnd.ms-powerpoint', 

785 'application/vnd.openxmlformats-officedocument.presentationml.presentation', 

786 ] or file_ext in ['ppt', 'pptx']: 

787 try: 

788 loader = UnstructuredLoader(file_path, 'ppt' if file_ext == 'ppt' else 'pptx') 

789 except ImportError: 

790 log.warning( 

791 "The 'unstructured' package is not installed. " 

792 'Falling back to python-pptx for PowerPoint file loading. ' 

793 'Install unstructured for better results: pip install unstructured' 

794 ) 

795 loader = PptxLoader(file_path) 

796 elif file_ext == 'msg': 796 ↛ 797line 796 didn't jump to line 797 because the condition on line 796 was never true

797 try: 

798 # unstructured parses .msg via python-oxmsg; avoids extract_msg's beautifulsoup4<4.14 conflict 

799 loader = UnstructuredLoader(file_path, 'msg', process_attachments=False) 

800 except ImportError: 

801 raise ValueError( 

802 "Processing .msg files requires the 'unstructured' package. " 

803 'Install it with: pip install unstructured' 

804 ) 

805 elif file_ext == 'odt': 805 ↛ 806line 805 didn't jump to line 806 because the condition on line 805 was never true

806 try: 

807 loader = UnstructuredLoader(file_path, 'odt') 

808 except ImportError: 

809 raise ValueError( 

810 "Processing .odt files requires the 'unstructured' package. " 

811 'Install it with: pip install unstructured' 

812 ) 

813 elif self._is_text_file(file_ext, file_content_type): 813 ↛ 814line 813 didn't jump to line 814 because the condition on line 813 was never true

814 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

815 else: 

816 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) 

817 

818 return loader