Coverage for open_webui/retrieval/loaders/youtube.py: 11%
103 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1import logging
2from typing import Any, Dict, Generator, List, Optional, Sequence, Union
3from urllib.parse import parse_qs, urlparse
4from xml.etree.ElementTree import ParseError
6from langchain_core.documents import Document
8log = logging.getLogger(__name__)
10ALLOWED_SCHEMES = {'http', 'https'}
11ALLOWED_NETLOCS = {
12 'youtu.be',
13 'm.youtube.com',
14 'youtube.com',
15 'www.youtube.com',
16 'www.youtube-nocookie.com',
17 'vid.plus',
18}
21class YoutubeTranscriptError(Exception):
22 """A YouTube transcript could not be retrieved."""
25def _transcript_error_message(error: Exception, video_id: str) -> str:
26 name = type(error).__name__
28 if name in {'RequestBlocked', 'IpBlocked'}:
29 return (
30 f'YouTube blocked the transcript request for {video_id} from this server. '
31 'This usually means the server address is rate limited or belongs to a cloud '
32 'provider. A proxy for these requests can be configured under Admin Settings, '
33 'Web Search, Youtube Proxy URL.'
34 )
35 if name == 'TranscriptsDisabled':
36 return f'Transcripts are disabled for the YouTube video {video_id}.'
37 if name == 'AgeRestricted':
38 return f'The YouTube video {video_id} is age restricted, so its transcript cannot be retrieved.'
39 if name in {'VideoUnavailable', 'VideoUnplayable', 'InvalidVideoId'}:
40 return f'The YouTube video {video_id} is unavailable.'
41 if name == 'PoTokenRequired':
42 return f'YouTube requires additional verification to return the transcript for {video_id}.'
44 return f'Could not retrieve a transcript for the YouTube video {video_id}.'
47def _parse_video_id(url: str) -> Optional[str]:
48 """Parse a YouTube URL and return the video ID if valid, otherwise None."""
49 parsed_url = urlparse(url)
51 if parsed_url.scheme not in ALLOWED_SCHEMES:
52 return None
54 if parsed_url.netloc not in ALLOWED_NETLOCS:
55 return None
57 path = parsed_url.path
59 if path.endswith('/watch'):
60 query = parsed_url.query
61 parsed_query = parse_qs(query)
62 if 'v' in parsed_query:
63 ids = parsed_query['v']
64 video_id = ids if isinstance(ids, str) else ids[0]
65 else:
66 return None
67 else:
68 path = parsed_url.path.lstrip('/')
69 video_id = path.split('/')[-1]
71 if len(video_id) != 11: # Video IDs are 11 characters long
72 return None
74 return video_id
77class YoutubeLoader:
78 """Load `YouTube` video transcripts."""
80 def __init__(
81 self,
82 video_id: str,
83 language: Union[str, Sequence[str]] = 'en',
84 proxy_url: Optional[str] = None,
85 ):
86 """Initialize with YouTube video ID."""
87 _video_id = _parse_video_id(video_id)
88 self.video_id = _video_id if _video_id is not None else video_id
89 self._metadata = {'source': video_id}
90 self.proxy_url = proxy_url
92 # Ensure language is a list
93 if isinstance(language, str):
94 self.language = [language]
95 else:
96 self.language = list(language)
98 # Add English as fallback if not already in the list
99 if 'en' not in self.language:
100 self.language.append('en')
102 def load(self) -> List[Document]:
103 """Load YouTube transcripts into `Document` objects."""
104 try:
105 from youtube_transcript_api import (
106 NoTranscriptFound,
107 TranscriptsDisabled,
108 YouTubeTranscriptApi,
109 )
110 from youtube_transcript_api.proxies import GenericProxyConfig
111 except ImportError:
112 raise ImportError(
113 'Could not import "youtube_transcript_api" Python package. '
114 'Please install it with `pip install youtube-transcript-api`.'
115 )
117 if self.proxy_url:
118 youtube_proxies = GenericProxyConfig(http_url=self.proxy_url, https_url=self.proxy_url)
119 log.debug('Using proxy URL: %s...', self.proxy_url[:14])
120 else:
121 youtube_proxies = None
123 transcript_api = YouTubeTranscriptApi(proxy_config=youtube_proxies)
124 try:
125 transcript_list = transcript_api.list(self.video_id)
126 except Exception as e:
127 log.warning('Loading YouTube transcript failed: %s', e)
128 raise YoutubeTranscriptError(_transcript_error_message(e, self.video_id)) from e
130 # Try each language in order of priority
131 for lang in self.language:
132 try:
133 transcript = transcript_list.find_transcript([lang])
134 if transcript.is_generated:
135 log.debug("Found generated transcript for language '%s'", lang)
136 try:
137 transcript = transcript_list.find_manually_created_transcript([lang])
138 log.debug("Found manual transcript for language '%s'", lang)
139 except NoTranscriptFound:
140 log.debug("No manual transcript found for language '%s', using generated", lang)
141 pass
143 log.debug("Found transcript for language '%s'", lang)
144 try:
145 transcript_pieces: List[Dict[str, Any]] = transcript.fetch()
146 except ParseError:
147 log.debug("Empty or invalid transcript for language '%s'", lang)
148 continue
150 if not transcript_pieces:
151 log.debug("Empty transcript for language '%s'", lang)
152 continue
154 transcript_text = ' '.join(
155 map(
156 lambda transcript_piece: (
157 transcript_piece.text.strip(' ') if hasattr(transcript_piece, 'text') else ''
158 ),
159 transcript_pieces,
160 )
161 )
162 return [Document(page_content=transcript_text, metadata=self._metadata)]
163 except NoTranscriptFound:
164 log.debug("No transcript found for language '%s'", lang)
165 continue
166 except Exception as e:
167 log.info("Error finding transcript for language '%s'", lang)
168 raise YoutubeTranscriptError(_transcript_error_message(e, self.video_id)) from e
170 # If we get here, all languages failed
171 languages_tried = ', '.join(self.language)
172 log.warning(
173 f'No transcript found for any of the specified languages: {languages_tried}. Verify if the video has transcripts, add more languages if needed.'
174 )
175 raise YoutubeTranscriptError(
176 f'No transcript found for the YouTube video {self.video_id} in these languages: {languages_tried}.'
177 )
179 async def aload(self) -> Generator[Document, None, None]:
180 """Asynchronously load YouTube transcripts into `Document` objects."""
181 import asyncio
183 loop = asyncio.get_event_loop()
184 return await loop.run_in_executor(None, self.load)