Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
youtube.py530 linesDownload Raw Back to document_loaders
1"""Loads YouTube transcript."""2 3from __future__ import annotations4 5import logging6from enum import Enum7from pathlib import Path8from typing import Any, Dict, Generator, List, Optional, Sequence, Union9from urllib.parse import parse_qs, urlparse10from xml.etree.ElementTree import ParseError  # OK: trusted-source11 12from langchain_core.documents import Document13from pydantic import model_validator14from pydantic.dataclasses import dataclass15 16from langchain_community.document_loaders.base import BaseLoader17 18logger = logging.getLogger(__name__)19 20SCOPES = ["https://www.googleapis.com/auth/youtube.readonly"]21 22 23@dataclass24class GoogleApiClient:25    """Generic Google API Client.26 27    To use, you should have the ``google_auth_oauthlib,youtube_transcript_api,google``28    python package installed.29    As the google api expects credentials you need to set up a google account and30    register your Service. "https://developers.google.com/docs/api/quickstart/python"31 32    *Security Note*: Note that parsing of the transcripts relies on the standard33        xml library but the input is viewed as trusted in this case.34 35 36    Example:37        .. code-block:: python38 39            from langchain_community.document_loaders import GoogleApiClient40            google_api_client = GoogleApiClient(41                service_account_path=Path("path_to_your_sec_file.json")42            )43 44    """45 46    credentials_path: Path = Path.home() / ".credentials" / "credentials.json"47    service_account_path: Path = Path.home() / ".credentials" / "credentials.json"48    token_path: Path = Path.home() / ".credentials" / "token.json"49 50    def __post_init__(self) -> None:51        self.creds = self._load_credentials()52 53    @model_validator(mode="before")54    @classmethod55    def validate_channel_or_videoIds_is_set(cls, values: Any) -> Any:56        """Validate that either folder_id or document_ids is set, but not both."""57 58        if not values.kwargs.get("credentials_path") and not values.kwargs.get(59            "service_account_path"60        ):61            raise ValueError("Must specify either channel_name or video_ids")62        return values.kwargs63 64    def _load_credentials(self) -> Any:65        """Load credentials."""66        # Adapted from https://developers.google.com/drive/api/v3/quickstart/python67        try:68            from google.auth.transport.requests import Request69            from google.oauth2 import service_account70            from google.oauth2.credentials import Credentials71            from google_auth_oauthlib.flow import InstalledAppFlow72            from youtube_transcript_api import YouTubeTranscriptApi  # noqa: F40173        except ImportError:74            raise ImportError(75                "You must run"76                "`pip install --upgrade "77                "google-api-python-client google-auth-httplib2 "78                "google-auth-oauthlib "79                "youtube-transcript-api` "80                "to use the Google Drive loader"81            )82 83        creds = None84        if self.service_account_path.exists():85            return service_account.Credentials.from_service_account_file(86                str(self.service_account_path)87            )88        if self.token_path.exists():89            creds = Credentials.from_authorized_user_file(str(self.token_path), SCOPES)90 91        if not creds or not creds.valid:92            if creds and creds.expired and creds.refresh_token:93                creds.refresh(Request())94            else:95                flow = InstalledAppFlow.from_client_secrets_file(96                    str(self.credentials_path), SCOPES97                )98                creds = flow.run_local_server(port=0)99            with open(self.token_path, "w") as token:100                token.write(creds.to_json())101 102        return creds103 104 105ALLOWED_SCHEMES = {"http", "https"}106ALLOWED_NETLOCS = {107    "youtu.be",108    "m.youtube.com",109    "youtube.com",110    "www.youtube.com",111    "www.youtube-nocookie.com",112    "vid.plus",113}114 115 116def _parse_video_id(url: str) -> Optional[str]:117    """Parse a YouTube URL and return the video ID if valid, otherwise None."""118    parsed_url = urlparse(url)119 120    if parsed_url.scheme not in ALLOWED_SCHEMES:121        return None122 123    if parsed_url.netloc not in ALLOWED_NETLOCS:124        return None125 126    path = parsed_url.path127 128    if path.endswith("/watch"):129        query = parsed_url.query130        parsed_query = parse_qs(query)131        if "v" in parsed_query:132            ids = parsed_query["v"]133            video_id = ids if isinstance(ids, str) else ids[0]134        else:135            return None136    else:137        path = parsed_url.path.lstrip("/")138        video_id = path.split("/")[-1]139 140    if len(video_id) != 11:  # Video IDs are 11 characters long141        return None142 143    return video_id144 145 146class TranscriptFormat(Enum):147    """Output formats of transcripts from `YoutubeLoader`."""148 149    TEXT = "text"150    LINES = "lines"151    CHUNKS = "chunks"152 153 154class YoutubeLoader(BaseLoader):155    """Load `YouTube` video transcripts."""156 157    def __init__(158        self,159        video_id: str,160        add_video_info: bool = False,161        language: Union[str, Sequence[str]] = "en",162        translation: Optional[str] = None,163        transcript_format: TranscriptFormat = TranscriptFormat.TEXT,164        continue_on_failure: bool = False,165        chunk_size_seconds: int = 120,166    ):167        """Initialize with YouTube video ID."""168        self.video_id = video_id169        self._metadata = {"source": video_id}170        self.add_video_info = add_video_info171        self.language = language172        if isinstance(language, str):173            self.language = [language]174        else:175            self.language = language176        self.translation = translation177        self.transcript_format = transcript_format178        self.continue_on_failure = continue_on_failure179        self.chunk_size_seconds = chunk_size_seconds180 181    @staticmethod182    def extract_video_id(youtube_url: str) -> str:183        """Extract video ID from common YouTube URLs."""184        video_id = _parse_video_id(youtube_url)185        if not video_id:186            raise ValueError(187                f'Could not determine the video ID for the URL "{youtube_url}".'188            )189        return video_id190 191    @classmethod192    def from_youtube_url(cls, youtube_url: str, **kwargs: Any) -> YoutubeLoader:193        """Given a YouTube URL, construct a loader.194        See `YoutubeLoader()` constructor for a list of keyword arguments.195        """196        video_id = cls.extract_video_id(youtube_url)197        return cls(video_id, **kwargs)198 199    def _make_chunk_document(200        self, chunk_pieces: List[Dict], chunk_start_seconds: int201    ) -> Document:202        """Create Document from chunk of transcript pieces."""203        m, s = divmod(chunk_start_seconds, 60)204        h, m = divmod(m, 60)205        return Document(206            page_content=" ".join(207                map(lambda chunk_piece: chunk_piece["text"].strip(" "), chunk_pieces)208            ),209            metadata={210                **self._metadata,211                "start_seconds": chunk_start_seconds,212                "start_timestamp": f"{h:02d}:{m:02d}:{s:02d}",213                "source":214                # replace video ID with URL to start time215                f"https://www.youtube.com/watch?v={self.video_id}"216                f"&t={chunk_start_seconds}s",217            },218        )219 220    def _get_transcript_chunks(221        self, transcript_pieces: List[Dict]222    ) -> Generator[Document, None, None]:223        chunk_pieces: List[Dict[str, Any]] = []224        chunk_start_seconds = 0225        chunk_time_limit = self.chunk_size_seconds226        for transcript_piece in transcript_pieces:227            piece_end = transcript_piece["start"] + transcript_piece["duration"]228            if piece_end > chunk_time_limit:229                if chunk_pieces:230                    yield self._make_chunk_document(chunk_pieces, chunk_start_seconds)231                chunk_pieces = []232                chunk_start_seconds = chunk_time_limit233                chunk_time_limit += self.chunk_size_seconds234 235            chunk_pieces.append(transcript_piece)236 237        if len(chunk_pieces) > 0:238            yield self._make_chunk_document(chunk_pieces, chunk_start_seconds)239 240    def load(self) -> List[Document]:241        """Load YouTube transcripts into `Document` objects."""242        try:243            from youtube_transcript_api import (244                FetchedTranscript,245                NoTranscriptFound,246                TranscriptsDisabled,247                YouTubeTranscriptApi,248            )249        except ImportError:250            raise ImportError(251                'Could not import "youtube_transcript_api" Python package. '252                "Please install it with `pip install youtube-transcript-api`."253            )254 255        if self.add_video_info:256            # Get more video meta info257            # Such as title, description, thumbnail url, publish_date258            video_info = self._get_video_info()259            self._metadata.update(video_info)260 261        try:262            ytt_api = YouTubeTranscriptApi()263            transcript_list = ytt_api.list(self.video_id)264        except TranscriptsDisabled:265            return []266 267        try:268            transcript = transcript_list.find_transcript(self.language)269        except NoTranscriptFound:270            transcript = transcript_list.find_transcript(["en"])271 272        if self.translation is not None:273            transcript = transcript.translate(self.translation)274        transcript_object = transcript.fetch()275        if isinstance(transcript_object, FetchedTranscript):276            transcript_pieces = [277                {278                    "text": snippet.text,279                    "start": snippet.start,280                    "duration": snippet.duration,281                }282                for snippet in transcript_object.snippets283            ]284        else:285            transcript_pieces: List[Dict[str, Any]] = transcript_object  # type: ignore[no-redef]286 287        if self.transcript_format == TranscriptFormat.TEXT:288            transcript = " ".join(289                map(290                    lambda transcript_piece: transcript_piece["text"].strip(" "),291                    transcript_pieces,292                )293            )294            return [Document(page_content=transcript, metadata=self._metadata)]295        elif self.transcript_format == TranscriptFormat.LINES:296            return list(297                map(298                    lambda transcript_piece: Document(299                        page_content=transcript_piece["text"].strip(" "),300                        metadata=dict(301                            filter(302                                lambda item: item[0] != "text", transcript_piece.items()303                            )304                        ),305                    ),306                    transcript_pieces,307                )308            )309        elif self.transcript_format == TranscriptFormat.CHUNKS:310            return list(self._get_transcript_chunks(transcript_pieces))311 312        else:313            raise ValueError("Unknown transcript format.")314 315    def _get_video_info(self) -> Dict:316        """Get important video information.317 318        Components include:319            - title320            - description321            - thumbnail URL,322            - publish_date323            - channel author324            - and more.325        """326        try:327            from pytube import YouTube328 329        except ImportError:330            raise ImportError(331                'Could not import "pytube" Python package. '332                "Please install it with `pip install pytube`."333            )334        yt = YouTube(f"https://www.youtube.com/watch?v={self.video_id}")335        video_info = {336            "title": yt.title or "Unknown",337            "description": yt.description or "Unknown",338            "view_count": yt.views or 0,339            "thumbnail_url": yt.thumbnail_url or "Unknown",340            "publish_date": yt.publish_date.strftime("%Y-%m-%d %H:%M:%S")341            if yt.publish_date342            else "Unknown",343            "length": yt.length or 0,344            "author": yt.author or "Unknown",345        }346        return video_info347 348 349@dataclass350class GoogleApiYoutubeLoader(BaseLoader):351    """Load all Videos from a `YouTube` Channel.352 353    To use, you should have the ``googleapiclient,youtube_transcript_api``354    python package installed.355    As the service needs a google_api_client, you first have to initialize356    the GoogleApiClient.357 358    Additionally you have to either provide a channel name or a list of videoids359    "https://developers.google.com/docs/api/quickstart/python"360 361 362 363    Example:364        .. code-block:: python365 366            from langchain_community.document_loaders import GoogleApiClient367            from langchain_community.document_loaders import GoogleApiYoutubeLoader368            google_api_client = GoogleApiClient(369                service_account_path=Path("path_to_your_sec_file.json")370            )371            loader = GoogleApiYoutubeLoader(372                google_api_client=google_api_client,373                channel_name = "CodeAesthetic"374            )375            load.load()376 377    """378 379    google_api_client: GoogleApiClient380    channel_name: Optional[str] = None381    video_ids: Optional[List[str]] = None382    add_video_info: bool = True383    captions_language: str = "en"384    continue_on_failure: bool = False385 386    def __post_init__(self) -> None:387        self.youtube_client = self._build_youtube_client(self.google_api_client.creds)388 389    def _build_youtube_client(self, creds: Any) -> Any:390        try:391            from googleapiclient.discovery import build392            from youtube_transcript_api import YouTubeTranscriptApi  # noqa: F401393        except ImportError:394            raise ImportError(395                "You must run"396                "`pip install --upgrade "397                "google-api-python-client google-auth-httplib2 "398                "google-auth-oauthlib "399                "youtube-transcript-api` "400                "to use the Google Drive loader"401            )402 403        return build("youtube", "v3", credentials=creds)404 405    @model_validator(mode="before")406    @classmethod407    def validate_channel_or_videoIds_is_set(cls, values: Any) -> Any:408        """Validate that either folder_id or document_ids is set, but not both."""409        if not values.kwargs.get("channel_name") and not values.kwargs.get("video_ids"):410            raise ValueError("Must specify either channel_name or video_ids")411        return values.kwargs412 413    def _get_transcripe_for_video_id(self, video_id: str) -> str:414        from youtube_transcript_api import NoTranscriptFound, YouTubeTranscriptApi415 416        ytt_api = YouTubeTranscriptApi()417        transcript_list = ytt_api.list(video_id)418        try:419            transcript = transcript_list.find_transcript([self.captions_language])420        except NoTranscriptFound:421            for available_transcript in transcript_list:422                transcript = available_transcript.translate(self.captions_language)423                continue424 425        transcript_pieces = transcript.fetch()426        return " ".join([t["text"].strip(" ") for t in transcript_pieces])427 428    def _get_document_for_video_id(self, video_id: str, **kwargs: Any) -> Document:429        captions = self._get_transcripe_for_video_id(video_id)430        video_response = (431            self.youtube_client.videos()432            .list(433                part="id,snippet",434                id=video_id,435            )436            .execute()437        )438        return Document(439            page_content=captions,440            metadata=video_response.get("items")[0],441        )442 443    def _get_channel_id(self, channel_name: str) -> str:444        request = self.youtube_client.search().list(445            part="id",446            q=channel_name,447            type="channel",448            maxResults=1,  # we only need one result since channel names are unique449        )450        response = request.execute()451        channel_id = response["items"][0]["id"]["channelId"]452        return channel_id453 454    def _get_uploads_playlist_id(self, channel_id: str) -> str:455        request = self.youtube_client.channels().list(456            part="contentDetails",457            id=channel_id,458        )459        response = request.execute()460        return response["items"][0]["contentDetails"]["relatedPlaylists"]["uploads"]461 462    def _get_document_for_channel(self, channel: str, **kwargs: Any) -> List[Document]:463        try:464            from youtube_transcript_api import (465                NoTranscriptFound,466                TranscriptsDisabled,467            )468        except ImportError:469            raise ImportError(470                "You must run"471                "`pip install --upgrade "472                "youtube-transcript-api` "473                "to use the youtube loader"474            )475 476        channel_id = self._get_channel_id(channel)477        uploads_playlist_id = self._get_uploads_playlist_id(channel_id)478        request = self.youtube_client.playlistItems().list(479            part="id,snippet",480            playlistId=uploads_playlist_id,481            maxResults=50,482        )483        video_ids = []484        while request is not None:485            response = request.execute()486 487            # Add each video ID to the list488            for item in response["items"]:489                video_id = item["snippet"]["resourceId"]["videoId"]490                meta_data = {"videoId": video_id}491                if self.add_video_info:492                    item["snippet"].pop("thumbnails")493                    meta_data.update(item["snippet"])494                try:495                    page_content = self._get_transcripe_for_video_id(video_id)496                    video_ids.append(497                        Document(498                            page_content=page_content,499                            metadata=meta_data,500                        )501                    )502                except (TranscriptsDisabled, NoTranscriptFound, ParseError) as e:503                    if self.continue_on_failure:504                        logger.error(505                            "Error fetching transscript "506                            + f" {item['id']['videoId']}, exception: {e}"507                        )508                    else:509                        raise e510                    pass511            request = self.youtube_client.search().list_next(request, response)512 513        return video_ids514 515    def load(self) -> List[Document]:516        """Load documents."""517        document_list = []518        if self.channel_name:519            document_list.extend(self._get_document_for_channel(self.channel_name))520        elif self.video_ids:521            document_list.extend(522                [523                    self._get_document_for_video_id(video_id)524                    for video_id in self.video_ids525                ]526            )527        else:528            raise ValueError("Must specify either channel_name or video_ids")529        return document_list530 
codekingpro/portable-devtools · Team Ai