Coverage for open_webui/retrieval/loaders/tavily.py: 18%
43 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1import logging
2from typing import Iterator, List, Literal, Union
4import requests
5from langchain_core.document_loaders import BaseLoader
6from langchain_core.documents import Document
7from open_webui.env import TAVILY_API_BASE_URL
9log = logging.getLogger(__name__)
12class TavilyLoader(BaseLoader):
13 """Extract web page content from URLs using Tavily Extract API.
15 This is a LangChain document loader that uses Tavily's Extract API to
16 retrieve content from web pages and return it as Document objects.
18 Args:
19 urls: URL or list of URLs to extract content from.
20 api_key: The Tavily API key.
21 extract_depth: Depth of extraction, either "basic" or "advanced".
22 continue_on_failure: Whether to continue if extraction of a URL fails.
23 """
25 def __init__(
26 self,
27 urls: Union[str, List[str]],
28 api_key: str,
29 extract_depth: Literal['basic', 'advanced'] = 'basic',
30 continue_on_failure: bool = True,
31 ) -> None:
32 """Initialize Tavily Extract client.
34 Args:
35 urls: URL or list of URLs to extract content from.
36 api_key: The Tavily API key.
37 include_images: Whether to include images in the extraction.
38 extract_depth: Depth of extraction, either "basic" or "advanced".
39 advanced extraction retrieves more data, including tables and
40 embedded content, with higher success but may increase latency.
41 basic costs 1 credit per 5 successful URL extractions,
42 advanced costs 2 credits per 5 successful URL extractions.
43 continue_on_failure: Whether to continue if extraction of a URL fails.
44 """
45 if not urls:
46 raise ValueError('At least one URL must be provided.')
48 self.api_key = api_key
49 self.urls = urls if isinstance(urls, list) else [urls]
50 self.extract_depth = extract_depth
51 self.continue_on_failure = continue_on_failure
52 self.api_url = f'{TAVILY_API_BASE_URL}/extract'
54 def lazy_load(self) -> Iterator[Document]:
55 """Extract and yield documents from the URLs using Tavily Extract API."""
56 batch_size = 20
57 for i in range(0, len(self.urls), batch_size):
58 batch_urls = self.urls[i : i + batch_size]
59 try:
60 headers = {
61 'Content-Type': 'application/json',
62 'Authorization': f'Bearer {self.api_key}',
63 }
64 # Use string for single URL, array for multiple URLs
65 urls_param = batch_urls[0] if len(batch_urls) == 1 else batch_urls
66 payload = {'urls': urls_param, 'extract_depth': self.extract_depth}
67 # Make the API call
68 response = requests.post(self.api_url, headers=headers, json=payload)
69 response.raise_for_status()
70 response_data = response.json()
71 # Process successful results
72 for result in response_data.get('results', []):
73 url = result.get('url', '')
74 content = result.get('raw_content', '')
75 if not content:
76 log.warning(f'No content extracted from {url}')
77 continue
78 # Add URLs as metadata
79 metadata = {'source': url}
80 yield Document(
81 page_content=content,
82 metadata=metadata,
83 )
84 for failed in response_data.get('failed_results', []):
85 url = failed.get('url', '')
86 error = failed.get('error', 'Unknown error')
87 log.error(f'Failed to extract content from {url}: {error}')
88 except Exception as e:
89 if self.continue_on_failure:
90 log.error(f'Error extracting content from batch {batch_urls}: {e}')
91 else:
92 raise e