codekingpro/portable-devtools
114k
1"""Loads YouTube transcript."""2 3from __future__ import annotations4 5import logging6from enum import Enum7from pathlib import Path8from typing import Any, Dict, Generator, List, Optional, Sequence, Union9from urllib.parse import parse_qs, urlparse10from xml.etree.ElementTree import ParseError # OK: trusted-source11 12from langchain_core.documents import Document13from pydantic import model_validator14from pydantic.dataclasses import dataclass15 16from langchain_community.document_loaders.base import BaseLoader17 18logger = logging.getLogger(__name__)19 20SCOPES = ["https://www.googleapis.com/auth/youtube.readonly"]21 22 23@dataclass24class GoogleApiClient:25 """Generic Google API Client.26 27 To use, you should have the ``google_auth_oauthlib,youtube_transcript_api,google``28 python package installed.29 As the google api expects credentials you need to set up a google account and30 register your Service. "https://developers.google.com/docs/api/quickstart/python"31 32 *Security Note*: Note that parsing of the transcripts relies on the standard33 xml library but the input is viewed as trusted in this case.34 35 36 Example:37 .. code-block:: python38 39 from langchain_community.document_loaders import GoogleApiClient40 google_api_client = GoogleApiClient(41 service_account_path=Path("path_to_your_sec_file.json")42 )43 44 """45 46 credentials_path: Path = Path.home() / ".credentials" / "credentials.json"47 service_account_path: Path = Path.home() / ".credentials" / "credentials.json"48 token_path: Path = Path.home() / ".credentials" / "token.json"49 50 def __post_init__(self) -> None:51 self.creds = self._load_credentials()52 53 @model_validator(mode="before")54 @classmethod55 def validate_channel_or_videoIds_is_set(cls, values: Any) -> Any:56 """Validate that either folder_id or document_ids is set, but not both."""57 58 if not values.kwargs.get("credentials_path") and not values.kwargs.get(59 "service_account_path"60 ):61 raise ValueError("Must specify either channel_name or video_ids")62 return values.kwargs63 64 def _load_credentials(self) -> Any:65 """Load credentials."""66 # Adapted from https://developers.google.com/drive/api/v3/quickstart/python67 try:68 from google.auth.transport.requests import Request69 from google.oauth2 import service_account70 from google.oauth2.credentials import Credentials71 from google_auth_oauthlib.flow import InstalledAppFlow72 from youtube_transcript_api import YouTubeTranscriptApi # noqa: F40173 except ImportError:74 raise ImportError(75 "You must run"76 "`pip install --upgrade "77 "google-api-python-client google-auth-httplib2 "78 "google-auth-oauthlib "79 "youtube-transcript-api` "80 "to use the Google Drive loader"81 )82 83 creds = None84 if self.service_account_path.exists():85 return service_account.Credentials.from_service_account_file(86 str(self.service_account_path)87 )88 if self.token_path.exists():89 creds = Credentials.from_authorized_user_file(str(self.token_path), SCOPES)90 91 if not creds or not creds.valid:92 if creds and creds.expired and creds.refresh_token:93 creds.refresh(Request())94 else:95 flow = InstalledAppFlow.from_client_secrets_file(96 str(self.credentials_path), SCOPES97 )98 creds = flow.run_local_server(port=0)99 with open(self.token_path, "w") as token:100 token.write(creds.to_json())101 102 return creds103 104 105ALLOWED_SCHEMES = {"http", "https"}106ALLOWED_NETLOCS = {107 "youtu.be",108 "m.youtube.com",109 "youtube.com",110 "www.youtube.com",111 "www.youtube-nocookie.com",112 "vid.plus",113}114 115 116def _parse_video_id(url: str) -> Optional[str]:117 """Parse a YouTube URL and return the video ID if valid, otherwise None."""118 parsed_url = urlparse(url)119 120 if parsed_url.scheme not in ALLOWED_SCHEMES:121 return None122 123 if parsed_url.netloc not in ALLOWED_NETLOCS:124 return None125 126 path = parsed_url.path127 128 if path.endswith("/watch"):129 query = parsed_url.query130 parsed_query = parse_qs(query)131 if "v" in parsed_query:132 ids = parsed_query["v"]133 video_id = ids if isinstance(ids, str) else ids[0]134 else:135 return None136 else:137 path = parsed_url.path.lstrip("/")138 video_id = path.split("/")[-1]139 140 if len(video_id) != 11: # Video IDs are 11 characters long141 return None142 143 return video_id144 145 146class TranscriptFormat(Enum):147 """Output formats of transcripts from `YoutubeLoader`."""148 149 TEXT = "text"150 LINES = "lines"151 CHUNKS = "chunks"152 153 154class YoutubeLoader(BaseLoader):155 """Load `YouTube` video transcripts."""156 157 def __init__(158 self,159 video_id: str,160 add_video_info: bool = False,161 language: Union[str, Sequence[str]] = "en",162 translation: Optional[str] = None,163 transcript_format: TranscriptFormat = TranscriptFormat.TEXT,164 continue_on_failure: bool = False,165 chunk_size_seconds: int = 120,166 ):167 """Initialize with YouTube video ID."""168 self.video_id = video_id169 self._metadata = {"source": video_id}170 self.add_video_info = add_video_info171 self.language = language172 if isinstance(language, str):173 self.language = [language]174 else:175 self.language = language176 self.translation = translation177 self.transcript_format = transcript_format178 self.continue_on_failure = continue_on_failure179 self.chunk_size_seconds = chunk_size_seconds180 181 @staticmethod182 def extract_video_id(youtube_url: str) -> str:183 """Extract video ID from common YouTube URLs."""184 video_id = _parse_video_id(youtube_url)185 if not video_id:186 raise ValueError(187 f'Could not determine the video ID for the URL "{youtube_url}".'188 )189 return video_id190 191 @classmethod192 def from_youtube_url(cls, youtube_url: str, **kwargs: Any) -> YoutubeLoader:193 """Given a YouTube URL, construct a loader.194 See `YoutubeLoader()` constructor for a list of keyword arguments.195 """196 video_id = cls.extract_video_id(youtube_url)197 return cls(video_id, **kwargs)198 199 def _make_chunk_document(200 self, chunk_pieces: List[Dict], chunk_start_seconds: int201 ) -> Document:202 """Create Document from chunk of transcript pieces."""203 m, s = divmod(chunk_start_seconds, 60)204 h, m = divmod(m, 60)205 return Document(206 page_content=" ".join(207 map(lambda chunk_piece: chunk_piece["text"].strip(" "), chunk_pieces)208 ),209 metadata={210 **self._metadata,211 "start_seconds": chunk_start_seconds,212 "start_timestamp": f"{h:02d}:{m:02d}:{s:02d}",213 "source":214 # replace video ID with URL to start time215 f"https://www.youtube.com/watch?v={self.video_id}"216 f"&t={chunk_start_seconds}s",217 },218 )219 220 def _get_transcript_chunks(221 self, transcript_pieces: List[Dict]222 ) -> Generator[Document, None, None]:223 chunk_pieces: List[Dict[str, Any]] = []224 chunk_start_seconds = 0225 chunk_time_limit = self.chunk_size_seconds226 for transcript_piece in transcript_pieces:227 piece_end = transcript_piece["start"] + transcript_piece["duration"]228 if piece_end > chunk_time_limit:229 if chunk_pieces:230 yield self._make_chunk_document(chunk_pieces, chunk_start_seconds)231 chunk_pieces = []232 chunk_start_seconds = chunk_time_limit233 chunk_time_limit += self.chunk_size_seconds234 235 chunk_pieces.append(transcript_piece)236 237 if len(chunk_pieces) > 0:238 yield self._make_chunk_document(chunk_pieces, chunk_start_seconds)239 240 def load(self) -> List[Document]:241 """Load YouTube transcripts into `Document` objects."""242 try:243 from youtube_transcript_api import (244 FetchedTranscript,245 NoTranscriptFound,246 TranscriptsDisabled,247 YouTubeTranscriptApi,248 )249 except ImportError:250 raise ImportError(251 'Could not import "youtube_transcript_api" Python package. '252 "Please install it with `pip install youtube-transcript-api`."253 )254 255 if self.add_video_info:256 # Get more video meta info257 # Such as title, description, thumbnail url, publish_date258 video_info = self._get_video_info()259 self._metadata.update(video_info)260 261 try:262 ytt_api = YouTubeTranscriptApi()263 transcript_list = ytt_api.list(self.video_id)264 except TranscriptsDisabled:265 return []266 267 try:268 transcript = transcript_list.find_transcript(self.language)269 except NoTranscriptFound:270 transcript = transcript_list.find_transcript(["en"])271 272 if self.translation is not None:273 transcript = transcript.translate(self.translation)274 transcript_object = transcript.fetch()275 if isinstance(transcript_object, FetchedTranscript):276 transcript_pieces = [277 {278 "text": snippet.text,279 "start": snippet.start,280 "duration": snippet.duration,281 }282 for snippet in transcript_object.snippets283 ]284 else:285 transcript_pieces: List[Dict[str, Any]] = transcript_object # type: ignore[no-redef]286 287 if self.transcript_format == TranscriptFormat.TEXT:288 transcript = " ".join(289 map(290 lambda transcript_piece: transcript_piece["text"].strip(" "),291 transcript_pieces,292 )293 )294 return [Document(page_content=transcript, metadata=self._metadata)]295 elif self.transcript_format == TranscriptFormat.LINES:296 return list(297 map(298 lambda transcript_piece: Document(299 page_content=transcript_piece["text"].strip(" "),300 metadata=dict(301 filter(302 lambda item: item[0] != "text", transcript_piece.items()303 )304 ),305 ),306 transcript_pieces,307 )308 )309 elif self.transcript_format == TranscriptFormat.CHUNKS:310 return list(self._get_transcript_chunks(transcript_pieces))311 312 else:313 raise ValueError("Unknown transcript format.")314 315 def _get_video_info(self) -> Dict:316 """Get important video information.317 318 Components include:319 - title320 - description321 - thumbnail URL,322 - publish_date323 - channel author324 - and more.325 """326 try:327 from pytube import YouTube328 329 except ImportError:330 raise ImportError(331 'Could not import "pytube" Python package. '332 "Please install it with `pip install pytube`."333 )334 yt = YouTube(f"https://www.youtube.com/watch?v={self.video_id}")335 video_info = {336 "title": yt.title or "Unknown",337 "description": yt.description or "Unknown",338 "view_count": yt.views or 0,339 "thumbnail_url": yt.thumbnail_url or "Unknown",340 "publish_date": yt.publish_date.strftime("%Y-%m-%d %H:%M:%S")341 if yt.publish_date342 else "Unknown",343 "length": yt.length or 0,344 "author": yt.author or "Unknown",345 }346 return video_info347 348 349@dataclass350class GoogleApiYoutubeLoader(BaseLoader):351 """Load all Videos from a `YouTube` Channel.352 353 To use, you should have the ``googleapiclient,youtube_transcript_api``354 python package installed.355 As the service needs a google_api_client, you first have to initialize356 the GoogleApiClient.357 358 Additionally you have to either provide a channel name or a list of videoids359 "https://developers.google.com/docs/api/quickstart/python"360 361 362 363 Example:364 .. code-block:: python365 366 from langchain_community.document_loaders import GoogleApiClient367 from langchain_community.document_loaders import GoogleApiYoutubeLoader368 google_api_client = GoogleApiClient(369 service_account_path=Path("path_to_your_sec_file.json")370 )371 loader = GoogleApiYoutubeLoader(372 google_api_client=google_api_client,373 channel_name = "CodeAesthetic"374 )375 load.load()376 377 """378 379 google_api_client: GoogleApiClient380 channel_name: Optional[str] = None381 video_ids: Optional[List[str]] = None382 add_video_info: bool = True383 captions_language: str = "en"384 continue_on_failure: bool = False385 386 def __post_init__(self) -> None:387 self.youtube_client = self._build_youtube_client(self.google_api_client.creds)388 389 def _build_youtube_client(self, creds: Any) -> Any:390 try:391 from googleapiclient.discovery import build392 from youtube_transcript_api import YouTubeTranscriptApi # noqa: F401393 except ImportError:394 raise ImportError(395 "You must run"396 "`pip install --upgrade "397 "google-api-python-client google-auth-httplib2 "398 "google-auth-oauthlib "399 "youtube-transcript-api` "400 "to use the Google Drive loader"401 )402 403 return build("youtube", "v3", credentials=creds)404 405 @model_validator(mode="before")406 @classmethod407 def validate_channel_or_videoIds_is_set(cls, values: Any) -> Any:408 """Validate that either folder_id or document_ids is set, but not both."""409 if not values.kwargs.get("channel_name") and not values.kwargs.get("video_ids"):410 raise ValueError("Must specify either channel_name or video_ids")411 return values.kwargs412 413 def _get_transcripe_for_video_id(self, video_id: str) -> str:414 from youtube_transcript_api import NoTranscriptFound, YouTubeTranscriptApi415 416 ytt_api = YouTubeTranscriptApi()417 transcript_list = ytt_api.list(video_id)418 try:419 transcript = transcript_list.find_transcript([self.captions_language])420 except NoTranscriptFound:421 for available_transcript in transcript_list:422 transcript = available_transcript.translate(self.captions_language)423 continue424 425 transcript_pieces = transcript.fetch()426 return " ".join([t["text"].strip(" ") for t in transcript_pieces])427 428 def _get_document_for_video_id(self, video_id: str, **kwargs: Any) -> Document:429 captions = self._get_transcripe_for_video_id(video_id)430 video_response = (431 self.youtube_client.videos()432 .list(433 part="id,snippet",434 id=video_id,435 )436 .execute()437 )438 return Document(439 page_content=captions,440 metadata=video_response.get("items")[0],441 )442 443 def _get_channel_id(self, channel_name: str) -> str:444 request = self.youtube_client.search().list(445 part="id",446 q=channel_name,447 type="channel",448 maxResults=1, # we only need one result since channel names are unique449 )450 response = request.execute()451 channel_id = response["items"][0]["id"]["channelId"]452 return channel_id453 454 def _get_uploads_playlist_id(self, channel_id: str) -> str:455 request = self.youtube_client.channels().list(456 part="contentDetails",457 id=channel_id,458 )459 response = request.execute()460 return response["items"][0]["contentDetails"]["relatedPlaylists"]["uploads"]461 462 def _get_document_for_channel(self, channel: str, **kwargs: Any) -> List[Document]:463 try:464 from youtube_transcript_api import (465 NoTranscriptFound,466 TranscriptsDisabled,467 )468 except ImportError:469 raise ImportError(470 "You must run"471 "`pip install --upgrade "472 "youtube-transcript-api` "473 "to use the youtube loader"474 )475 476 channel_id = self._get_channel_id(channel)477 uploads_playlist_id = self._get_uploads_playlist_id(channel_id)478 request = self.youtube_client.playlistItems().list(479 part="id,snippet",480 playlistId=uploads_playlist_id,481 maxResults=50,482 )483 video_ids = []484 while request is not None:485 response = request.execute()486 487 # Add each video ID to the list488 for item in response["items"]:489 video_id = item["snippet"]["resourceId"]["videoId"]490 meta_data = {"videoId": video_id}491 if self.add_video_info:492 item["snippet"].pop("thumbnails")493 meta_data.update(item["snippet"])494 try:495 page_content = self._get_transcripe_for_video_id(video_id)496 video_ids.append(497 Document(498 page_content=page_content,499 metadata=meta_data,500 )501 )502 except (TranscriptsDisabled, NoTranscriptFound, ParseError) as e:503 if self.continue_on_failure:504 logger.error(505 "Error fetching transscript "506 + f" {item['id']['videoId']}, exception: {e}"507 )508 else:509 raise e510 pass511 request = self.youtube_client.search().list_next(request, response)512 513 return video_ids514 515 def load(self) -> List[Document]:516 """Load documents."""517 document_list = []518 if self.channel_name:519 document_list.extend(self._get_document_for_channel(self.channel_name))520 elif self.video_ids:521 document_list.extend(522 [523 self._get_document_for_video_id(video_id)524 for video_id in self.video_ids525 ]526 )527 else:528 raise ValueError("Must specify either channel_name or video_ids")529 return document_list530 