Coverage for open_webui/retrieval/loaders/paddleocr_vl.py: 20%

61 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1import base64 

2import logging 

3import os 

4import sys 

5from typing import List 

6 

7import requests 

8from langchain_core.documents import Document 

9from open_webui.env import GLOBAL_LOG_LEVEL 

10 

11logging.basicConfig(stream=sys.stdout, level=GLOBAL_LOG_LEVEL) 

12log = logging.getLogger(__name__) 

13 

14PADDLEOCR_VL_IMAGE_EXTENSIONS = ['png', 'jpg', 'jpeg', 'bmp', 'tiff', 'webp'] 

15PADDLEOCR_VL_SUPPORTED_EXTENSIONS = ['pdf'] + PADDLEOCR_VL_IMAGE_EXTENSIONS 

16 

17 

18class PaddleOCRVLLoader: 

19 """Loader that uses PaddleOCR-vl API to extract text from PDF/images.""" 

20 

21 def __init__( 

22 self, 

23 api_url: str, 

24 token: str, 

25 file_path: str, 

26 ): 

27 if not api_url or not token: 

28 raise ValueError('PaddleOCR-vl API URL and Token are required.') 

29 if not os.path.exists(file_path): 

30 raise FileNotFoundError(f'File not found at {file_path}') 

31 

32 self.api_url = api_url.rstrip('/') 

33 self.token = token 

34 self.file_path = file_path 

35 self.file_name = os.path.basename(file_path) 

36 

37 def load(self) -> List[Document]: 

38 log.info('Processing with PaddleOCR-vl: %s', self.file_path) 

39 

40 try: 

41 with open(self.file_path, 'rb') as file: 

42 file_bytes = file.read() 

43 file_data = base64.b64encode(file_bytes).decode('ascii') 

44 except Exception as e: 

45 log.error(f'Failed to read file {self.file_path}: {e}') 

46 raise 

47 

48 headers = {'Authorization': f'token {self.token}', 'Content-Type': 'application/json'} 

49 

50 # Detect fileType based on file extension 

51 ext = self.file_path.lower().split('.')[-1] 

52 file_type = 1 if ext in PADDLEOCR_VL_IMAGE_EXTENSIONS else 0 

53 

54 payload = { 

55 'file': file_data, 

56 'fileType': file_type, 

57 'useDocOrientationClassify': False, 

58 'useDocUnwarping': False, 

59 'useChartRecognition': False, 

60 } 

61 

62 try: 

63 response = requests.post(f'{self.api_url}/layout-parsing', json=payload, headers=headers) 

64 response.raise_for_status() 

65 

66 result = response.json().get('result', {}) 

67 layout_results = result.get('layoutParsingResults', []) 

68 

69 documents = [] 

70 total_pages = len(layout_results) 

71 skipped_pages = 0 

72 

73 for i, res in enumerate(layout_results): 

74 markdown_text = res.get('markdown', {}).get('text', '') 

75 

76 if isinstance(markdown_text, str): 

77 cleaned_content = markdown_text.strip() 

78 else: 

79 cleaned_content = str(markdown_text).strip() 

80 

81 if not cleaned_content: 

82 skipped_pages += 1 

83 continue 

84 

85 documents.append( 

86 Document( 

87 page_content=cleaned_content, 

88 metadata={ 

89 'page': i, 

90 'page_label': i + 1, 

91 'total_pages': total_pages, 

92 'file_name': self.file_name, 

93 'processing_engine': 'paddleocr-vl', 

94 }, 

95 ) 

96 ) 

97 

98 if skipped_pages > 0: 

99 log.info('PaddleOCR-vl: Processed %s pages, skipped %s empty pages.', len(documents), skipped_pages) 

100 

101 if not documents: 

102 log.warning('No valid text content found by PaddleOCR-vl.') 

103 return [ 

104 Document( 

105 page_content='No valid text content found in document', 

106 metadata={ 

107 'error': 'no_valid_pages', 

108 'file_name': self.file_name, 

109 'processing_engine': 'paddleocr-vl', 

110 }, 

111 ) 

112 ] 

113 

114 return documents 

115 

116 except Exception as e: 

117 log.error(f'Error calling PaddleOCR-vl: {e}') 

118 return [ 

119 Document( 

120 page_content=f'Error during OCR processing: {e}', 

121 metadata={ 

122 'error': 'processing_failed', 

123 'file_name': self.file_name, 

124 'processing_engine': 'paddleocr-vl', 

125 }, 

126 ) 

127 ]