Coverage for open_webui/retrieval/loaders/paddleocr_vl.py: 20%
61 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1import base64
2import logging
3import os
4import sys
5from typing import List
7import requests
8from langchain_core.documents import Document
9from open_webui.env import GLOBAL_LOG_LEVEL
11logging.basicConfig(stream=sys.stdout, level=GLOBAL_LOG_LEVEL)
12log = logging.getLogger(__name__)
14PADDLEOCR_VL_IMAGE_EXTENSIONS = ['png', 'jpg', 'jpeg', 'bmp', 'tiff', 'webp']
15PADDLEOCR_VL_SUPPORTED_EXTENSIONS = ['pdf'] + PADDLEOCR_VL_IMAGE_EXTENSIONS
18class PaddleOCRVLLoader:
19 """Loader that uses PaddleOCR-vl API to extract text from PDF/images."""
21 def __init__(
22 self,
23 api_url: str,
24 token: str,
25 file_path: str,
26 ):
27 if not api_url or not token:
28 raise ValueError('PaddleOCR-vl API URL and Token are required.')
29 if not os.path.exists(file_path):
30 raise FileNotFoundError(f'File not found at {file_path}')
32 self.api_url = api_url.rstrip('/')
33 self.token = token
34 self.file_path = file_path
35 self.file_name = os.path.basename(file_path)
37 def load(self) -> List[Document]:
38 log.info('Processing with PaddleOCR-vl: %s', self.file_path)
40 try:
41 with open(self.file_path, 'rb') as file:
42 file_bytes = file.read()
43 file_data = base64.b64encode(file_bytes).decode('ascii')
44 except Exception as e:
45 log.error(f'Failed to read file {self.file_path}: {e}')
46 raise
48 headers = {'Authorization': f'token {self.token}', 'Content-Type': 'application/json'}
50 # Detect fileType based on file extension
51 ext = self.file_path.lower().split('.')[-1]
52 file_type = 1 if ext in PADDLEOCR_VL_IMAGE_EXTENSIONS else 0
54 payload = {
55 'file': file_data,
56 'fileType': file_type,
57 'useDocOrientationClassify': False,
58 'useDocUnwarping': False,
59 'useChartRecognition': False,
60 }
62 try:
63 response = requests.post(f'{self.api_url}/layout-parsing', json=payload, headers=headers)
64 response.raise_for_status()
66 result = response.json().get('result', {})
67 layout_results = result.get('layoutParsingResults', [])
69 documents = []
70 total_pages = len(layout_results)
71 skipped_pages = 0
73 for i, res in enumerate(layout_results):
74 markdown_text = res.get('markdown', {}).get('text', '')
76 if isinstance(markdown_text, str):
77 cleaned_content = markdown_text.strip()
78 else:
79 cleaned_content = str(markdown_text).strip()
81 if not cleaned_content:
82 skipped_pages += 1
83 continue
85 documents.append(
86 Document(
87 page_content=cleaned_content,
88 metadata={
89 'page': i,
90 'page_label': i + 1,
91 'total_pages': total_pages,
92 'file_name': self.file_name,
93 'processing_engine': 'paddleocr-vl',
94 },
95 )
96 )
98 if skipped_pages > 0:
99 log.info('PaddleOCR-vl: Processed %s pages, skipped %s empty pages.', len(documents), skipped_pages)
101 if not documents:
102 log.warning('No valid text content found by PaddleOCR-vl.')
103 return [
104 Document(
105 page_content='No valid text content found in document',
106 metadata={
107 'error': 'no_valid_pages',
108 'file_name': self.file_name,
109 'processing_engine': 'paddleocr-vl',
110 },
111 )
112 ]
114 return documents
116 except Exception as e:
117 log.error(f'Error calling PaddleOCR-vl: {e}')
118 return [
119 Document(
120 page_content=f'Error during OCR processing: {e}',
121 metadata={
122 'error': 'processing_failed',
123 'file_name': self.file_name,
124 'processing_engine': 'paddleocr-vl',
125 },
126 )
127 ]