Coverage for open_webui/retrieval/loaders/youtube.py: 11%

103 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1import logging 

2from typing import Any, Dict, Generator, List, Optional, Sequence, Union 

3from urllib.parse import parse_qs, urlparse 

4from xml.etree.ElementTree import ParseError 

5 

6from langchain_core.documents import Document 

7 

8log = logging.getLogger(__name__) 

9 

10ALLOWED_SCHEMES = {'http', 'https'} 

11ALLOWED_NETLOCS = { 

12 'youtu.be', 

13 'm.youtube.com', 

14 'youtube.com', 

15 'www.youtube.com', 

16 'www.youtube-nocookie.com', 

17 'vid.plus', 

18} 

19 

20 

21class YoutubeTranscriptError(Exception): 

22 """A YouTube transcript could not be retrieved.""" 

23 

24 

25def _transcript_error_message(error: Exception, video_id: str) -> str: 

26 name = type(error).__name__ 

27 

28 if name in {'RequestBlocked', 'IpBlocked'}: 

29 return ( 

30 f'YouTube blocked the transcript request for {video_id} from this server. ' 

31 'This usually means the server address is rate limited or belongs to a cloud ' 

32 'provider. A proxy for these requests can be configured under Admin Settings, ' 

33 'Web Search, Youtube Proxy URL.' 

34 ) 

35 if name == 'TranscriptsDisabled': 

36 return f'Transcripts are disabled for the YouTube video {video_id}.' 

37 if name == 'AgeRestricted': 

38 return f'The YouTube video {video_id} is age restricted, so its transcript cannot be retrieved.' 

39 if name in {'VideoUnavailable', 'VideoUnplayable', 'InvalidVideoId'}: 

40 return f'The YouTube video {video_id} is unavailable.' 

41 if name == 'PoTokenRequired': 

42 return f'YouTube requires additional verification to return the transcript for {video_id}.' 

43 

44 return f'Could not retrieve a transcript for the YouTube video {video_id}.' 

45 

46 

47def _parse_video_id(url: str) -> Optional[str]: 

48 """Parse a YouTube URL and return the video ID if valid, otherwise None.""" 

49 parsed_url = urlparse(url) 

50 

51 if parsed_url.scheme not in ALLOWED_SCHEMES: 

52 return None 

53 

54 if parsed_url.netloc not in ALLOWED_NETLOCS: 

55 return None 

56 

57 path = parsed_url.path 

58 

59 if path.endswith('/watch'): 

60 query = parsed_url.query 

61 parsed_query = parse_qs(query) 

62 if 'v' in parsed_query: 

63 ids = parsed_query['v'] 

64 video_id = ids if isinstance(ids, str) else ids[0] 

65 else: 

66 return None 

67 else: 

68 path = parsed_url.path.lstrip('/') 

69 video_id = path.split('/')[-1] 

70 

71 if len(video_id) != 11: # Video IDs are 11 characters long 

72 return None 

73 

74 return video_id 

75 

76 

77class YoutubeLoader: 

78 """Load `YouTube` video transcripts.""" 

79 

80 def __init__( 

81 self, 

82 video_id: str, 

83 language: Union[str, Sequence[str]] = 'en', 

84 proxy_url: Optional[str] = None, 

85 ): 

86 """Initialize with YouTube video ID.""" 

87 _video_id = _parse_video_id(video_id) 

88 self.video_id = _video_id if _video_id is not None else video_id 

89 self._metadata = {'source': video_id} 

90 self.proxy_url = proxy_url 

91 

92 # Ensure language is a list 

93 if isinstance(language, str): 

94 self.language = [language] 

95 else: 

96 self.language = list(language) 

97 

98 # Add English as fallback if not already in the list 

99 if 'en' not in self.language: 

100 self.language.append('en') 

101 

102 def load(self) -> List[Document]: 

103 """Load YouTube transcripts into `Document` objects.""" 

104 try: 

105 from youtube_transcript_api import ( 

106 NoTranscriptFound, 

107 TranscriptsDisabled, 

108 YouTubeTranscriptApi, 

109 ) 

110 from youtube_transcript_api.proxies import GenericProxyConfig 

111 except ImportError: 

112 raise ImportError( 

113 'Could not import "youtube_transcript_api" Python package. ' 

114 'Please install it with `pip install youtube-transcript-api`.' 

115 ) 

116 

117 if self.proxy_url: 

118 youtube_proxies = GenericProxyConfig(http_url=self.proxy_url, https_url=self.proxy_url) 

119 log.debug('Using proxy URL: %s...', self.proxy_url[:14]) 

120 else: 

121 youtube_proxies = None 

122 

123 transcript_api = YouTubeTranscriptApi(proxy_config=youtube_proxies) 

124 try: 

125 transcript_list = transcript_api.list(self.video_id) 

126 except Exception as e: 

127 log.warning('Loading YouTube transcript failed: %s', e) 

128 raise YoutubeTranscriptError(_transcript_error_message(e, self.video_id)) from e 

129 

130 # Try each language in order of priority 

131 for lang in self.language: 

132 try: 

133 transcript = transcript_list.find_transcript([lang]) 

134 if transcript.is_generated: 

135 log.debug("Found generated transcript for language '%s'", lang) 

136 try: 

137 transcript = transcript_list.find_manually_created_transcript([lang]) 

138 log.debug("Found manual transcript for language '%s'", lang) 

139 except NoTranscriptFound: 

140 log.debug("No manual transcript found for language '%s', using generated", lang) 

141 pass 

142 

143 log.debug("Found transcript for language '%s'", lang) 

144 try: 

145 transcript_pieces: List[Dict[str, Any]] = transcript.fetch() 

146 except ParseError: 

147 log.debug("Empty or invalid transcript for language '%s'", lang) 

148 continue 

149 

150 if not transcript_pieces: 

151 log.debug("Empty transcript for language '%s'", lang) 

152 continue 

153 

154 transcript_text = ' '.join( 

155 map( 

156 lambda transcript_piece: ( 

157 transcript_piece.text.strip(' ') if hasattr(transcript_piece, 'text') else '' 

158 ), 

159 transcript_pieces, 

160 ) 

161 ) 

162 return [Document(page_content=transcript_text, metadata=self._metadata)] 

163 except NoTranscriptFound: 

164 log.debug("No transcript found for language '%s'", lang) 

165 continue 

166 except Exception as e: 

167 log.info("Error finding transcript for language '%s'", lang) 

168 raise YoutubeTranscriptError(_transcript_error_message(e, self.video_id)) from e 

169 

170 # If we get here, all languages failed 

171 languages_tried = ', '.join(self.language) 

172 log.warning( 

173 f'No transcript found for any of the specified languages: {languages_tried}. Verify if the video has transcripts, add more languages if needed.' 

174 ) 

175 raise YoutubeTranscriptError( 

176 f'No transcript found for the YouTube video {self.video_id} in these languages: {languages_tried}.' 

177 ) 

178 

179 async def aload(self) -> Generator[Document, None, None]: 

180 """Asynchronously load YouTube transcripts into `Document` objects.""" 

181 import asyncio 

182 

183 loop = asyncio.get_event_loop() 

184 return await loop.run_in_executor(None, self.load)