Coverage for open_webui/retrieval/loaders/tavily.py: 18%

43 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1import logging 

2from typing import Iterator, List, Literal, Union 

3 

4import requests 

5from langchain_core.document_loaders import BaseLoader 

6from langchain_core.documents import Document 

7from open_webui.env import TAVILY_API_BASE_URL 

8 

9log = logging.getLogger(__name__) 

10 

11 

12class TavilyLoader(BaseLoader): 

13 """Extract web page content from URLs using Tavily Extract API. 

14 

15 This is a LangChain document loader that uses Tavily's Extract API to 

16 retrieve content from web pages and return it as Document objects. 

17 

18 Args: 

19 urls: URL or list of URLs to extract content from. 

20 api_key: The Tavily API key. 

21 extract_depth: Depth of extraction, either "basic" or "advanced". 

22 continue_on_failure: Whether to continue if extraction of a URL fails. 

23 """ 

24 

25 def __init__( 

26 self, 

27 urls: Union[str, List[str]], 

28 api_key: str, 

29 extract_depth: Literal['basic', 'advanced'] = 'basic', 

30 continue_on_failure: bool = True, 

31 ) -> None: 

32 """Initialize Tavily Extract client. 

33 

34 Args: 

35 urls: URL or list of URLs to extract content from. 

36 api_key: The Tavily API key. 

37 include_images: Whether to include images in the extraction. 

38 extract_depth: Depth of extraction, either "basic" or "advanced". 

39 advanced extraction retrieves more data, including tables and 

40 embedded content, with higher success but may increase latency. 

41 basic costs 1 credit per 5 successful URL extractions, 

42 advanced costs 2 credits per 5 successful URL extractions. 

43 continue_on_failure: Whether to continue if extraction of a URL fails. 

44 """ 

45 if not urls: 

46 raise ValueError('At least one URL must be provided.') 

47 

48 self.api_key = api_key 

49 self.urls = urls if isinstance(urls, list) else [urls] 

50 self.extract_depth = extract_depth 

51 self.continue_on_failure = continue_on_failure 

52 self.api_url = f'{TAVILY_API_BASE_URL}/extract' 

53 

54 def lazy_load(self) -> Iterator[Document]: 

55 """Extract and yield documents from the URLs using Tavily Extract API.""" 

56 batch_size = 20 

57 for i in range(0, len(self.urls), batch_size): 

58 batch_urls = self.urls[i : i + batch_size] 

59 try: 

60 headers = { 

61 'Content-Type': 'application/json', 

62 'Authorization': f'Bearer {self.api_key}', 

63 } 

64 # Use string for single URL, array for multiple URLs 

65 urls_param = batch_urls[0] if len(batch_urls) == 1 else batch_urls 

66 payload = {'urls': urls_param, 'extract_depth': self.extract_depth} 

67 # Make the API call 

68 response = requests.post(self.api_url, headers=headers, json=payload) 

69 response.raise_for_status() 

70 response_data = response.json() 

71 # Process successful results 

72 for result in response_data.get('results', []): 

73 url = result.get('url', '') 

74 content = result.get('raw_content', '') 

75 if not content: 

76 log.warning(f'No content extracted from {url}') 

77 continue 

78 # Add URLs as metadata 

79 metadata = {'source': url} 

80 yield Document( 

81 page_content=content, 

82 metadata=metadata, 

83 ) 

84 for failed in response_data.get('failed_results', []): 

85 url = failed.get('url', '') 

86 error = failed.get('error', 'Unknown error') 

87 log.error(f'Failed to extract content from {url}: {error}') 

88 except Exception as e: 

89 if self.continue_on_failure: 

90 log.error(f'Error extracting content from batch {batch_urls}: {e}') 

91 else: 

92 raise e