Coverage for open_webui/retrieval/loaders/main.py: 24%
366 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1import asyncio
2import csv
3import logging
4import os
5import sys
6import zipfile
8import ftfy
9import requests
10from fastapi import HTTPException
11from azure.identity import DefaultAzureCredential
12from langchain_core.documents import Document
13from open_webui.env import (
14 AIOHTTP_CLIENT_SESSION_SSL,
15 GLOBAL_LOG_LEVEL,
16 USE_SLIM,
17 MINERU_MAX_MARKDOWN_BYTES,
18 REQUESTS_VERIFY,
19)
20from open_webui.retrieval.loaders.datalab_marker import DatalabMarkerLoader
21from open_webui.retrieval.loaders.external_document import ExternalDocumentLoader
22from open_webui.retrieval.loaders.local import (
23 DocumentIntelligenceLoader,
24 DocxLoader,
25 HTMLLoader,
26 TextLoader,
27 UnstructuredLoader,
28)
29from open_webui.retrieval.loaders.mineru import MinerULoader
30from open_webui.retrieval.loaders.mistral import MistralLoader
31from open_webui.retrieval.loaders.paddleocr_vl import PADDLEOCR_VL_SUPPORTED_EXTENSIONS, PaddleOCRVLLoader
32from open_webui.retrieval.loaders.pdf import PDFLoader
33from open_webui.utils.headers import get_user_groups_for_custom_headers
34from open_webui.utils.json_codec import JSONCodec
36logging.basicConfig(stream=sys.stdout, level=GLOBAL_LOG_LEVEL)
37log = logging.getLogger(__name__)
39known_source_ext = [
40 'go',
41 'py',
42 'java',
43 'sh',
44 'bat',
45 'ps1',
46 'cmd',
47 'js',
48 'ts',
49 'css',
50 'cpp',
51 'hpp',
52 'h',
53 'c',
54 'cs',
55 'ino',
56 'sql',
57 'log',
58 'ini',
59 'pl',
60 'pm',
61 'r',
62 'dart',
63 'dockerfile',
64 'env',
65 'php',
66 'hs',
67 'hsc',
68 'lua',
69 'nginxconf',
70 'conf',
71 'm',
72 'mm',
73 'plsql',
74 'perl',
75 'rb',
76 'rs',
77 'db2',
78 'scala',
79 'bash',
80 'swift',
81 'vue',
82 'svelte',
83 'ex',
84 'exs',
85 'erl',
86 'tsx',
87 'jsx',
88 'hs',
89 'lhs',
90 'json',
91 'yaml',
92 'yml',
93 'toml',
94 'svg',
95]
97known_archive_ext = {'docx', 'epub', 'odt', 'pptx', 'xlsx'}
98known_archive_content_types = {
99 'application/epub+zip',
100 'application/vnd.oasis.opendocument.text',
101 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
102 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet',
103 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
104}
107class ExcelLoader:
108 """Fallback Excel loader using pandas when unstructured is not installed."""
110 def __init__(self, file_path):
111 self.file_path = file_path
113 def load(self) -> list[Document]:
114 import pandas as pd
116 text_parts = []
117 xls = pd.ExcelFile(self.file_path)
118 for sheet_name in xls.sheet_names:
119 df = pd.read_excel(xls, sheet_name=sheet_name)
120 text_parts.append(f'Sheet: {sheet_name}\n{df.to_string(index=False)}')
121 return [
122 Document(
123 page_content='\n\n'.join(text_parts),
124 metadata={'source': self.file_path},
125 )
126 ]
129def get_csv_summary(filename: str, file_path: str, encoding: str) -> str | None:
130 try:
131 with open(file_path, newline='', encoding=encoding) as f:
132 sample = f.read(4096)
133 f.seek(0)
134 try:
135 dialect = csv.Sniffer().sniff(sample)
136 except csv.Error:
137 dialect = csv.excel
139 total_rows = 0
140 max_columns = 0
141 headers = []
142 for row in csv.reader(f, dialect):
143 total_rows += 1
144 max_columns = max(max_columns, len(row))
145 if total_rows == 1:
146 headers = [header.lstrip('\ufeff') for header in row]
147 except Exception:
148 return None
150 if total_rows == 0:
151 return None
153 return (
154 f'Table: {total_rows} rows incl. header; '
155 f'{max(total_rows - 1, 0)} data rows; '
156 f'{max_columns} columns: {", ".join(headers)}.'
157 )
160class CSVLoaderWithSummary:
161 def __init__(self, file_path: str, filename: str, encoding: str):
162 self.file_path = file_path
163 self.filename = filename
164 self.encoding = encoding
166 def load(self) -> list[Document]:
167 docs = []
168 try:
169 with open(self.file_path, newline='', encoding=self.encoding) as file:
170 for index, row in enumerate(csv.DictReader(file)):
171 fields = []
172 for key, value in row.items():
173 if isinstance(value, str):
174 value = value.strip()
175 elif isinstance(value, list):
176 value = ','.join(v.strip() for v in value)
177 fields.append(f'{key.strip() if key is not None else key}: {value}')
178 content = '\n'.join(fields)
179 docs.append(Document(page_content=content, metadata={'source': self.file_path, 'row': index}))
180 except Exception as e:
181 raise RuntimeError(f'Error loading {self.file_path}') from e
182 if os.getenv('ENABLE_RAG_CSV_SUMMARY', 'False').lower() == 'true':
183 summary = get_csv_summary(self.filename, self.file_path, self.encoding)
184 if summary:
185 docs.insert(0, Document(page_content=summary, metadata={'source': self.file_path, 'row': -1}))
186 return docs
189class PptxLoader:
190 """Fallback PowerPoint loader using python-pptx when unstructured is not installed."""
192 def __init__(self, file_path):
193 self.file_path = file_path
195 def load(self) -> list[Document]:
196 from pptx import Presentation
198 prs = Presentation(self.file_path)
199 text_parts = []
200 for i, slide in enumerate(prs.slides, 1):
201 slide_texts = []
202 for shape in slide.shapes:
203 if shape.has_text_frame:
204 slide_texts.append(shape.text_frame.text)
205 if slide_texts:
206 text_parts.append(f'Slide {i}:\n' + '\n'.join(slide_texts))
207 return [
208 Document(
209 page_content='\n\n'.join(text_parts),
210 metadata={'source': self.file_path},
211 )
212 ]
215class TikaLoader:
216 def __init__(self, url, file_path, mime_type=None, extract_images=None, server_version='3'):
217 self.url = url
218 self.file_path = file_path
219 self.mime_type = mime_type
220 self.server_version = str(server_version or '3')
222 self.extract_images = extract_images
224 def load(self) -> list[Document]:
225 with open(self.file_path, 'rb') as f:
226 data = f.read()
228 if self.mime_type is not None:
229 headers = {'Content-Type': self.mime_type}
230 else:
231 headers = {}
233 if self.extract_images == True:
234 headers['X-Tika-PDFextractInlineImages'] = 'true'
236 endpoint_path = 'tika/json/md' if self.server_version == '4' else 'tika/text'
237 content_key = 'tk:content' if self.server_version == '4' else 'X-TIKA:content'
238 endpoint = f'{self.url.rstrip("/")}/{endpoint_path}'
240 r = requests.put(endpoint, data=data, headers=headers, verify=REQUESTS_VERIFY)
242 if r.ok:
243 raw_metadata = r.json()
244 text = raw_metadata.get(content_key, '<No text content found>').strip()
246 if 'Content-Type' in raw_metadata:
247 headers['Content-Type'] = raw_metadata['Content-Type']
249 log.debug('Tika extracted text: %s', text)
251 return [Document(page_content=text, metadata=headers)]
252 else:
253 raise Exception(f'Error calling Tika: {r.reason}')
256class DoclingLoader:
257 def __init__(self, url, api_key=None, file_path=None, mime_type=None, params=None):
258 self.url = url.rstrip('/')
259 self.api_key = api_key
260 self.file_path = file_path
261 self.mime_type = mime_type
263 self.params = params or {}
265 def load(self) -> list[Document]:
266 page_break_marker = '\f'
267 with open(self.file_path, 'rb') as f:
268 headers = {}
269 if self.api_key:
270 headers['X-Api-Key'] = f'{self.api_key}'
272 r = requests.post(
273 f'{self.url}/v1/convert/file',
274 files={
275 'files': (
276 self.file_path,
277 f,
278 self.mime_type or 'application/octet-stream',
279 )
280 },
281 data={
282 'image_export_mode': 'placeholder',
283 'md_page_break_placeholder': page_break_marker,
284 # Keep Docling params as user-provided form values. Encoding nested
285 # values here would make Open WebUI responsible for Docling's API
286 # quirks and could break when Docling changes its form contract.
287 **self.params,
288 },
289 headers=headers,
290 verify=AIOHTTP_CLIENT_SESSION_SSL,
291 )
292 if r.ok:
293 result = r.json()
294 # Docling reports failed and skipped conversions inside HTTP 200 responses.
295 conversion_status = result.get('status')
296 if conversion_status in ['failure', 'skipped']:
297 error_details = (
298 '; '.join(filter(None, (error.get('error_message') for error in result.get('errors', []))))
299 or 'no error message provided'
300 )
301 raise Exception(f'Error calling Docling: conversion status {conversion_status} - {error_details}')
303 document_data = result.get('document', {})
304 md_content = document_data.get('md_content') or ''
305 text = md_content or '<No text content found>'
307 metadata = {'Content-Type': self.mime_type} if self.mime_type else {}
308 if page_break_marker in md_content:
309 documents = [
310 Document(page_content=page.strip(), metadata={**metadata, 'page': page_idx})
311 for page_idx, page in enumerate(md_content.split(page_break_marker))
312 if page.strip()
313 ]
314 if documents:
315 log.debug('Docling extracted text: %s', text)
316 return documents
318 log.debug('Docling extracted text: %s', text)
319 return [Document(page_content=text, metadata=metadata)]
320 else:
321 error_msg = f'Error calling Docling API: {r.reason}'
322 if r.text:
323 try:
324 error_data = r.json()
325 if 'detail' in error_data:
326 error_msg += f' - {error_data["detail"]}'
327 except Exception:
328 error_msg += f' - {r.text}'
329 raise Exception(f'Error calling Docling: {error_msg}')
332class Loader:
333 def __init__(self, engine: str = '', **kwargs):
334 self.engine = engine
335 self.user = kwargs.get('user', None)
336 self.user_groups = kwargs.get('user_groups', None)
337 self.metadata = kwargs.get('metadata', {})
338 self.kwargs = kwargs
340 def load(self, filename: str, file_content_type: str, file_path: str) -> list[Document]:
341 loader = self._get_loader(filename, file_content_type, file_path)
342 docs = loader.load()
343 # ftfy's auto mode unescapes entities on every line before the first literal '<', rewriting the document.
344 return [
345 Document(page_content=ftfy.fix_text(doc.page_content, unescape_html=False), metadata=doc.metadata)
346 for doc in docs
347 ]
349 async def aload(self, filename: str, file_content_type: str, file_path: str) -> list[Document]:
350 """
351 Async wrapper around `load`.
353 Document loaders dispatched by `_get_loader` (PyMuPDF, Unstructured,
354 python-docx, Tika, etc.) are uniformly synchronous and CPU/IO-bound.
355 Calling `load` directly from an async handler would block the event
356 loop for the entire parse — minutes for large PDFs. This offloads
357 the work to a worker thread so the loop stays responsive.
358 """
359 # Group lookup is async-only, so it must happen before `load`
360 # is offloaded to a thread without a running event loop.
361 if self.engine == 'external' and self.user_groups is None: 361 ↛ 362line 361 didn't jump to line 362 because the condition on line 361 was never true
362 self.user_groups = await get_user_groups_for_custom_headers(
363 self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_HEADERS'), self.user
364 )
366 return await asyncio.to_thread(self.load, filename, file_content_type, file_path)
368 def _is_text_file(self, file_ext: str, file_content_type: str) -> bool:
369 return file_ext in known_source_ext or (
370 file_content_type
371 and file_content_type.find('text/') >= 0
372 # Avoid text/html files being detected as text
373 and not file_content_type.find('html') >= 0
374 )
376 def _detect_text_encoding(self, file_path: str) -> str:
377 """Detect the encoding of a text file with CJK-aware fallbacks.
379 Langchain's ``TextLoader`` uses chardet internally when
380 ``autodetect_encoding=True``, but chardet frequently misidentifies
381 CJK encodings (e.g. GB18030 detected as GB2312 or even Cyrillic).
382 This method replaces that by:
384 1. Trying UTF-8 first (fast path for the vast majority of files).
385 2. Using chardet as a *hint* to prioritise the right CJK codec
386 family, but mapping subset names to their superset
387 (e.g. GB2312 → gb18030).
388 3. Validating that decoded text actually contains CJK characters,
389 guarding against codecs that "succeed" but produce garbage.
390 4. Falling back to latin-1 (always valid, ftfy fixes mojibake later).
391 """
392 try:
393 with open(file_path, 'rb') as f:
394 raw = f.read()
395 except OSError:
396 return 'utf-8'
398 if not raw: 398 ↛ 399line 398 didn't jump to line 399 because the condition on line 398 was never true
399 return 'utf-8'
401 # Fast path: most files are UTF-8
402 try:
403 raw.decode('utf-8')
404 return 'utf-8'
405 except UnicodeDecodeError as e:
406 first_non_utf8 = e.start
408 # Use chardet as a hint, not as ground truth
409 import chardet
411 # chardet is pure Python (~1.3s/MB), so sample around the first bad byte
412 window = 256 * 1024
413 sample_start = max(0, first_non_utf8 - window // 2)
414 sample = raw[sample_start : sample_start + window]
415 detected = chardet.detect(sample)
416 # A stray byte can sit far from the real payload, leaving the sample with nothing to read
417 if len(sample.translate(None, delete=bytes(range(128)))) < 64 and len(sample) < len(raw):
418 detected = chardet.detect(raw)
419 detected_enc = (detected.get('encoding') or '').lower().replace('-', '').replace('_', '')
421 # Map chardet's detected encoding to the correct superset codec.
422 # chardet often reports GB2312 for content that is actually GB18030;
423 # GB18030 is a strict superset of both GB2312 and GBK.
424 _ENC_FAMILY = {
425 'gb2312': 'gb18030',
426 'gb18030': 'gb18030',
427 'gbk': 'gb18030',
428 'big5': 'big5',
429 'euckr': 'euc-kr',
430 'eucjp': 'euc-jp',
431 'iso2022jp': 'euc-jp',
432 'shiftjis': 'shift_jis',
433 }
435 # Build priority list: chardet-hinted codec first, then remaining CJK
436 base_order = ['gb18030', 'big5', 'euc-kr', 'euc-jp']
437 hinted = _ENC_FAMILY.get(detected_enc)
438 if hinted and hinted in base_order:
439 ordered = [hinted] + [e for e in base_order if e != hinted]
440 else:
441 ordered = base_order
443 for enc in ordered:
444 try:
445 text = raw.decode(enc)
446 if text.strip() and self._has_cjk_characters(text):
447 log.info(
448 'Detected encoding %s for %s (chardet guessed %s)',
449 enc,
450 file_path,
451 detected.get('encoding'),
452 )
453 return enc
454 except (UnicodeDecodeError, LookupError):
455 continue
457 # If chardet gave a non-CJK answer that isn't in our family map,
458 # try it directly — it might be a valid Western encoding.
459 chardet_encoding = detected.get('encoding')
460 if chardet_encoding:
461 try:
462 raw.decode(chardet_encoding)
463 log.info(
464 'Using chardet-detected encoding %s for %s',
465 chardet_encoding,
466 file_path,
467 )
468 return chardet_encoding
469 except (UnicodeDecodeError, LookupError):
470 pass
472 # latin-1 is the ultimate fallback: every byte 0x00–0xFF is valid.
473 # ftfy.fix_text() (applied downstream) repairs most mojibake that
474 # results from treating Windows-1252 content as Latin-1.
475 log.info('Falling back to latin-1 encoding for %s', file_path)
476 return 'latin-1'
478 @staticmethod
479 def _has_cjk_characters(text: str, threshold: float = 0.05) -> bool:
480 """Check if decoded text contains a meaningful proportion of CJK characters.
482 This guards against codecs that technically "succeed" but decode the
483 bytes into wrong Unicode codepoints (e.g. PUA chars, random symbols).
484 A genuine CJK document should have at least ``threshold`` fraction of
485 its non-whitespace characters in CJK Unicode blocks.
486 """
487 if not text:
488 return False
490 cjk_count = 0
491 total = 0
492 for ch in text:
493 if ch.isspace():
494 continue
495 total += 1
496 cp = ord(ch)
497 if (
498 0x4E00 <= cp <= 0x9FFF # CJK Unified Ideographs
499 or 0x3400 <= cp <= 0x4DBF # CJK Extension A
500 or 0x20000 <= cp <= 0x2A6DF # CJK Extension B
501 or 0x2A700 <= cp <= 0x2B73F # CJK Extension C
502 or 0x2B740 <= cp <= 0x2B81F # CJK Extension D
503 or 0xF900 <= cp <= 0xFAFF # CJK Compatibility Ideographs
504 or 0x3000 <= cp <= 0x303F # CJK Symbols and Punctuation
505 or 0x3040 <= cp <= 0x309F # Hiragana
506 or 0x30A0 <= cp <= 0x30FF # Katakana
507 or 0xAC00 <= cp <= 0xD7AF # Hangul Syllables
508 or 0xFF00 <= cp <= 0xFFEF # Halfwidth and Fullwidth Forms
509 ):
510 cjk_count += 1
512 if total == 0:
513 return False
515 return (cjk_count / total) >= threshold
517 def _get_loader(self, filename: str, file_content_type: str, file_path: str):
518 file_ext = filename.split('.')[-1].lower()
520 if file_ext in known_archive_ext or file_content_type in known_archive_content_types: 520 ↛ 521line 520 didn't jump to line 521 because the condition on line 520 was never true
521 max_file_size = self.kwargs.get('FILE_MAX_SIZE')
522 try:
523 max_file_size_bytes = int(max_file_size) * 1024 * 1024 if max_file_size else 100 * 1024 * 1024
524 except (TypeError, ValueError):
525 max_file_size_bytes = 100 * 1024 * 1024
527 if max_file_size_bytes > 0:
528 try:
529 with zipfile.ZipFile(file_path) as archive:
530 uncompressed_size = sum(entry.file_size for entry in archive.infolist())
531 except (zipfile.BadZipFile, OSError):
532 pass
533 else:
534 max_bytes = min(
535 max(10 * 1024 * 1024, os.path.getsize(file_path) * 100),
536 max_file_size_bytes,
537 )
538 if uncompressed_size > max_bytes:
539 raise ValueError('Document archive is too large after decompression')
541 if ( 541 ↛ 546line 541 didn't jump to line 546 because the condition on line 541 was never true
542 self.engine == 'external'
543 and self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_URL')
544 and self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_API_KEY')
545 ):
546 loader = ExternalDocumentLoader(
547 file_path=file_path,
548 url=self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_URL'),
549 api_key=self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_API_KEY'),
550 mime_type=file_content_type,
551 user=self.user,
552 user_groups=self.user_groups,
553 headers=self.kwargs.get('EXTERNAL_DOCUMENT_LOADER_HEADERS'),
554 metadata={
555 **self.metadata,
556 'file_name': filename,
557 'file_content_type': file_content_type,
558 },
559 )
560 elif self.engine == 'tika' and self.kwargs.get('TIKA_SERVER_URL'): 560 ↛ 561line 560 didn't jump to line 561 because the condition on line 560 was never true
561 if self._is_text_file(file_ext, file_content_type):
562 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
563 else:
564 loader = TikaLoader(
565 url=self.kwargs.get('TIKA_SERVER_URL'),
566 file_path=file_path,
567 server_version=self.kwargs.get('TIKA_SERVER_VERSION'),
568 extract_images=self.kwargs.get('PDF_EXTRACT_IMAGES'),
569 )
570 elif ( 570 ↛ 595line 570 didn't jump to line 595 because the condition on line 570 was never true
571 self.engine == 'datalab_marker'
572 and self.kwargs.get('DATALAB_MARKER_API_KEY')
573 and file_ext
574 in [
575 'pdf',
576 'xls',
577 'xlsx',
578 'ods',
579 'doc',
580 'docx',
581 'odt',
582 'ppt',
583 'pptx',
584 'odp',
585 'html',
586 'epub',
587 'png',
588 'jpeg',
589 'jpg',
590 'webp',
591 'gif',
592 'tiff',
593 ]
594 ):
595 api_base_url = self.kwargs.get('DATALAB_MARKER_API_BASE_URL', '')
596 if not api_base_url or api_base_url.strip() == '':
597 api_base_url = 'https://www.datalab.to/api/v1/marker' # https://github.com/open-webui/open-webui/pull/16867#issuecomment-3218424349
599 loader = DatalabMarkerLoader(
600 file_path=file_path,
601 api_key=self.kwargs['DATALAB_MARKER_API_KEY'],
602 api_base_url=api_base_url,
603 additional_config=self.kwargs.get('DATALAB_MARKER_ADDITIONAL_CONFIG'),
604 use_llm=self.kwargs.get('DATALAB_MARKER_USE_LLM', False),
605 skip_cache=self.kwargs.get('DATALAB_MARKER_SKIP_CACHE', False),
606 force_ocr=self.kwargs.get('DATALAB_MARKER_FORCE_OCR', False),
607 paginate=self.kwargs.get('DATALAB_MARKER_PAGINATE', False),
608 strip_existing_ocr=self.kwargs.get('DATALAB_MARKER_STRIP_EXISTING_OCR', False),
609 disable_image_extraction=self.kwargs.get('DATALAB_MARKER_DISABLE_IMAGE_EXTRACTION', False),
610 format_lines=self.kwargs.get('DATALAB_MARKER_FORMAT_LINES', False),
611 output_format=self.kwargs.get('DATALAB_MARKER_OUTPUT_FORMAT', 'markdown'),
612 )
613 elif self.engine == 'docling' and self.kwargs.get('DOCLING_SERVER_URL'): 613 ↛ 614line 613 didn't jump to line 614 because the condition on line 613 was never true
614 if self._is_text_file(file_ext, file_content_type):
615 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
616 else:
617 # Build params for DoclingLoader
618 params = self.kwargs.get('DOCLING_PARAMS', {})
619 if not isinstance(params, dict):
620 try:
621 params = JSONCodec.loads(params)
622 except JSONCodec.JSONDecodeError:
623 log.error('Invalid DOCLING_PARAMS format, expected JSON object')
624 params = {}
626 loader = DoclingLoader(
627 url=self.kwargs.get('DOCLING_SERVER_URL'),
628 api_key=self.kwargs.get('DOCLING_API_KEY', None),
629 file_path=file_path,
630 mime_type=file_content_type,
631 params=params,
632 )
633 elif ( 633 ↛ 646line 633 didn't jump to line 646 because the condition on line 633 was never true
634 self.engine == 'document_intelligence'
635 and self.kwargs.get('DOCUMENT_INTELLIGENCE_ENDPOINT') != ''
636 and (
637 file_ext in ['pdf', 'docx', 'ppt', 'pptx']
638 or file_content_type
639 in [
640 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
641 'application/vnd.ms-powerpoint',
642 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
643 ]
644 )
645 ):
646 if self.kwargs.get('DOCUMENT_INTELLIGENCE_KEY') != '':
647 loader = DocumentIntelligenceLoader(
648 file_path=file_path,
649 api_endpoint=self.kwargs.get('DOCUMENT_INTELLIGENCE_ENDPOINT'),
650 api_key=self.kwargs.get('DOCUMENT_INTELLIGENCE_KEY'),
651 api_model=self.kwargs.get('DOCUMENT_INTELLIGENCE_MODEL'),
652 )
653 else:
654 loader = DocumentIntelligenceLoader(
655 file_path=file_path,
656 api_endpoint=self.kwargs.get('DOCUMENT_INTELLIGENCE_ENDPOINT'),
657 azure_credential=DefaultAzureCredential(),
658 api_model=self.kwargs.get('DOCUMENT_INTELLIGENCE_MODEL'),
659 )
660 elif self.engine == 'mineru' and file_ext in self.kwargs.get('MINERU_FILE_EXTENSIONS', ['pdf']): 660 ↛ 661line 660 didn't jump to line 661 because the condition on line 660 was never true
661 mineru_timeout = self.kwargs.get('MINERU_API_TIMEOUT', 300)
662 if mineru_timeout:
663 try:
664 mineru_timeout = int(mineru_timeout)
665 except ValueError:
666 mineru_timeout = 300
667 loader = MinerULoader(
668 file_path=file_path,
669 api_mode=self.kwargs.get('MINERU_API_MODE', 'local'),
670 api_url=self.kwargs.get('MINERU_API_URL', 'http://localhost:8000'),
671 api_key=self.kwargs.get('MINERU_API_KEY', ''),
672 params=self.kwargs.get('MINERU_PARAMS', {}),
673 timeout=mineru_timeout,
674 max_markdown_bytes=MINERU_MAX_MARKDOWN_BYTES,
675 )
676 elif ( 676 ↛ 681line 676 didn't jump to line 681 because the condition on line 676 was never true
677 self.engine == 'mistral_ocr'
678 and self.kwargs.get('MISTRAL_OCR_API_KEY') != ''
679 and file_ext in ['pdf'] # Mistral OCR currently only supports PDF and images
680 ):
681 loader = MistralLoader(
682 base_url=self.kwargs.get('MISTRAL_OCR_API_BASE_URL'),
683 api_key=self.kwargs.get('MISTRAL_OCR_API_KEY'),
684 file_path=file_path,
685 use_base64=self.kwargs.get('MISTRAL_OCR_USE_BASE64', False),
686 user=self.user,
687 )
688 elif ( 688 ↛ 694line 688 didn't jump to line 694 because the condition on line 688 was never true
689 self.engine == 'paddleocr_vl'
690 and self.kwargs.get('PADDLEOCR_VL_BASE_URL')
691 and self.kwargs.get('PADDLEOCR_VL_TOKEN')
692 and file_ext in PADDLEOCR_VL_SUPPORTED_EXTENSIONS
693 ):
694 loader = PaddleOCRVLLoader(
695 api_url=self.kwargs.get('PADDLEOCR_VL_BASE_URL'),
696 token=self.kwargs.get('PADDLEOCR_VL_TOKEN'),
697 file_path=file_path,
698 )
699 else:
700 if USE_SLIM: 700 ↛ 701line 700 didn't jump to line 701 because the condition on line 700 was never true
701 if file_ext == 'csv':
702 return CSVLoaderWithSummary(file_path, filename, self._detect_text_encoding(file_path))
703 if file_ext in ['htm', 'html']:
704 return HTMLLoader(file_path, encoding=self._detect_text_encoding(file_path))
705 if file_ext in ['txt', 'md', 'markdown', 'rst', 'xml'] or self._is_text_file(
706 file_ext, file_content_type
707 ):
708 return TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
709 raise HTTPException(
710 503,
711 'This file type requires an external document extractor in slim. Configure one that supports it.',
712 )
713 if file_ext == 'pdf': 713 ↛ 714line 713 didn't jump to line 714 because the condition on line 713 was never true
714 loader = PDFLoader(
715 file_path,
716 extract_images=self.kwargs.get('PDF_EXTRACT_IMAGES'),
717 mode=self.kwargs.get('PDF_LOADER_MODE', 'page'),
718 )
719 elif file_ext == 'csv': 719 ↛ 720line 719 didn't jump to line 720 because the condition on line 719 was never true
720 loader = CSVLoaderWithSummary(
721 file_path,
722 filename,
723 self._detect_text_encoding(file_path),
724 )
725 elif file_ext == 'rst': 725 ↛ 726line 725 didn't jump to line 726 because the condition on line 725 was never true
726 try:
727 loader = UnstructuredLoader(file_path, 'rst', mode='elements')
728 except ImportError:
729 log.warning(
730 "The 'unstructured' package is not installed. "
731 'Falling back to plain text loading for .rst file. '
732 'Install it with: pip install unstructured'
733 )
734 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
735 elif file_ext == 'xml': 735 ↛ 736line 735 didn't jump to line 736 because the condition on line 735 was never true
736 try:
737 loader = UnstructuredLoader(file_path, 'xml')
738 except ImportError:
739 log.warning(
740 "The 'unstructured' package is not installed. "
741 'Falling back to plain text loading for .xml file. '
742 'Install it with: pip install unstructured'
743 )
744 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
745 elif file_ext in ['htm', 'html']: 745 ↛ 746line 745 didn't jump to line 746 because the condition on line 745 was never true
746 loader = HTMLLoader(file_path, encoding='unicode_escape')
747 elif file_ext == 'md': 747 ↛ 748line 747 didn't jump to line 748 because the condition on line 747 was never true
748 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
749 elif file_content_type == 'application/epub+zip': 749 ↛ 750line 749 didn't jump to line 750 because the condition on line 749 was never true
750 try:
751 loader = UnstructuredLoader(file_path, 'epub')
752 except ImportError:
753 raise ValueError(
754 "Processing .epub files requires the 'unstructured' package. "
755 'Install it with: pip install unstructured'
756 )
757 elif ( 757 ↛ 761line 757 didn't jump to line 761 because the condition on line 757 was never true
758 file_content_type == 'application/vnd.openxmlformats-officedocument.wordprocessingml.document'
759 or file_ext == 'docx'
760 ):
761 loader = DocxLoader(file_path)
762 elif file_ext == 'doc' or file_content_type == 'application/msword': 762 ↛ 763line 762 didn't jump to line 763 because the condition on line 762 was never true
763 try:
764 loader = UnstructuredLoader(file_path, 'doc')
765 except ImportError:
766 raise ValueError(
767 "Processing .doc files requires the 'unstructured' package. "
768 'Install it with: pip install unstructured'
769 )
770 elif file_content_type in [ 770 ↛ 774line 770 didn't jump to line 774 because the condition on line 770 was never true
771 'application/vnd.ms-excel',
772 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet',
773 ] or file_ext in ['xls', 'xlsx']:
774 try:
775 loader = UnstructuredLoader(file_path, 'xlsx')
776 except ImportError:
777 log.warning(
778 "The 'unstructured' package is not installed. "
779 'Falling back to pandas for Excel file loading. '
780 'Install unstructured for better results: pip install unstructured'
781 )
782 loader = ExcelLoader(file_path)
783 elif file_content_type in [ 783 ↛ 787line 783 didn't jump to line 787 because the condition on line 783 was never true
784 'application/vnd.ms-powerpoint',
785 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
786 ] or file_ext in ['ppt', 'pptx']:
787 try:
788 loader = UnstructuredLoader(file_path, 'ppt' if file_ext == 'ppt' else 'pptx')
789 except ImportError:
790 log.warning(
791 "The 'unstructured' package is not installed. "
792 'Falling back to python-pptx for PowerPoint file loading. '
793 'Install unstructured for better results: pip install unstructured'
794 )
795 loader = PptxLoader(file_path)
796 elif file_ext == 'msg': 796 ↛ 797line 796 didn't jump to line 797 because the condition on line 796 was never true
797 try:
798 # unstructured parses .msg via python-oxmsg; avoids extract_msg's beautifulsoup4<4.14 conflict
799 loader = UnstructuredLoader(file_path, 'msg', process_attachments=False)
800 except ImportError:
801 raise ValueError(
802 "Processing .msg files requires the 'unstructured' package. "
803 'Install it with: pip install unstructured'
804 )
805 elif file_ext == 'odt': 805 ↛ 806line 805 didn't jump to line 806 because the condition on line 805 was never true
806 try:
807 loader = UnstructuredLoader(file_path, 'odt')
808 except ImportError:
809 raise ValueError(
810 "Processing .odt files requires the 'unstructured' package. "
811 'Install it with: pip install unstructured'
812 )
813 elif self._is_text_file(file_ext, file_content_type): 813 ↛ 814line 813 didn't jump to line 814 because the condition on line 813 was never true
814 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
815 else:
816 loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path))
818 return loader