Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
blackboard.py303 linesDownload Raw Back to document_loaders
1import contextlib2import re3from pathlib import Path4from typing import Any, List, Optional, Tuple5from urllib.parse import unquote6 7from langchain_core.documents import Document8 9from langchain_community.document_loaders.directory import DirectoryLoader10from langchain_community.document_loaders.pdf import PyPDFLoader11from langchain_community.document_loaders.web_base import WebBaseLoader12 13 14class BlackboardLoader(WebBaseLoader):15    """Load a `Blackboard` course.16 17    This loader is not compatible with all Blackboard courses. It is only18    compatible with courses that use the new Blackboard interface.19    To use this loader, you must have the BbRouter cookie. You can get this20    cookie by logging into the course and then copying the value of the21    BbRouter cookie from the browser's developer tools.22 23    Example:24        .. code-block:: python25 26            from langchain_community.document_loaders import BlackboardLoader27 28            loader = BlackboardLoader(29                blackboard_course_url="https://blackboard.example.com/webapps/blackboard/execute/announcement?method=search&context=course_entry&course_id=_123456_1",30                bbrouter="expires:12345...",31            )32            documents = loader.load()33 34    """35 36    def __init__(37        self,38        blackboard_course_url: str,39        bbrouter: str,40        load_all_recursively: bool = True,41        basic_auth: Optional[Tuple[str, str]] = None,42        cookies: Optional[dict] = None,43        continue_on_failure: bool = False,44        show_progress: bool = True,45    ):46        """Initialize with blackboard course url.47 48        The BbRouter cookie is required for most blackboard courses.49 50        Args:51            blackboard_course_url: Blackboard course url.52            bbrouter: BbRouter cookie.53            load_all_recursively: If True, load all documents recursively.54            basic_auth: Basic auth credentials.55            cookies: Cookies.56            continue_on_failure: whether to continue loading the sitemap if an error57                occurs loading a url, emitting a warning instead of raising an58                exception. Setting this to True makes the loader more robust, but also59                may result in missing data. Default: False60            show_progress: whether to show a progress bar while loading. Default: True61 62        Raises:63            ValueError: If blackboard course url is invalid.64        """65        super().__init__(66            web_paths=(blackboard_course_url),67            continue_on_failure=continue_on_failure,68            show_progress=show_progress,69        )70        # Get base url71        try:72            self.base_url = blackboard_course_url.split("/webapps/blackboard")[0]73        except IndexError:74            raise IndexError(75                "Invalid blackboard course url. "76                "Please provide a url that starts with "77                "https://<blackboard_url>/webapps/blackboard"78            )79        if basic_auth is not None:80            self.session.auth = basic_auth81        # Combine cookies82        if cookies is None:83            cookies = {}84        cookies.update({"BbRouter": bbrouter})85        self.session.cookies.update(cookies)86        self.load_all_recursively = load_all_recursively87        self.check_bs4()88 89    def check_bs4(self) -> None:90        """Check if BeautifulSoup4 is installed.91 92        Raises:93            ImportError: If BeautifulSoup4 is not installed.94        """95        try:96            import bs4  # noqa: F40197        except ImportError:98            raise ImportError(99                "BeautifulSoup4 is required for BlackboardLoader. "100                "Please install it with `pip install beautifulsoup4`."101            )102 103    def load(self) -> List[Document]:104        """Load data into Document objects.105 106        Returns:107            List of Documents.108        """109        if self.load_all_recursively:110            soup_info = self.scrape()111            self.folder_path = self._get_folder_path(soup_info)112            relative_paths = self._get_paths(soup_info)113            documents = []114            for path in relative_paths:115                url = self.base_url + path116                print(f"Fetching documents from {url}")  # noqa: T201117                soup_info = self._scrape(url)118                with contextlib.suppress(ValueError):119                    documents.extend(self._get_documents(soup_info))120            return documents121        else:122            print(f"Fetching documents from {self.web_path}")  # noqa: T201123            soup_info = self.scrape()124            self.folder_path = self._get_folder_path(soup_info)125            return self._get_documents(soup_info)126 127    def _get_folder_path(self, soup: Any) -> str:128        """Get the folder path to save the Documents in.129 130        Args:131            soup: BeautifulSoup4 soup object.132 133        Returns:134            Folder path.135        """136        # Get the course name137        course_name = soup.find("span", {"id": "crumb_1"})138        if course_name is None:139            raise ValueError("No course name found.")140        course_name = course_name.text.strip()141        # Prepare the folder path142        course_name_clean = (143            unquote(course_name)144            .replace(" ", "_")145            .replace("/", "_")146            .replace(":", "_")147            .replace(",", "_")148            .replace("?", "_")149            .replace("'", "_")150            .replace("!", "_")151            .replace('"', "_")152        )153        # Get the folder path154        folder_path = Path(".") / course_name_clean155        return str(folder_path)156 157    def _get_documents(self, soup: Any) -> List[Document]:158        """Fetch content from page and return Documents.159 160        Args:161            soup: BeautifulSoup4 soup object.162 163        Returns:164            List of documents.165        """166        attachments = self._get_attachments(soup)167        self._download_attachments(attachments)168        documents = self._load_documents()169        return documents170 171    def _get_attachments(self, soup: Any) -> List[str]:172        """Get all attachments from a page.173 174        Args:175            soup: BeautifulSoup4 soup object.176 177        Returns:178            List of attachments.179        """180        from bs4 import BeautifulSoup, Tag181 182        # Get content list183        content_list: BeautifulSoup184        content_list = soup.find("ul", {"class": "contentList"})185        if content_list is None:186            raise ValueError("No content list found.")187        # Get all attachments188        attachments = []189        attachment: Tag190        for attachment in content_list.find_all("ul", {"class": "attachments"}):191            link: Tag192            for link in attachment.find_all("a"):193                href = link.get("href")194                # Only add if href is not None and does not start with #195                if href is not None and not href.startswith("#"):  # type: ignore[union-attr]196                    attachments.append(href)197        return attachments  # type: ignore[return-value]198 199    def _download_attachments(self, attachments: List[str]) -> None:200        """Download all attachments.201 202        Args:203            attachments: List of attachments.204        """205        # Make sure the folder exists206        Path(self.folder_path).mkdir(parents=True, exist_ok=True)207        # Download all attachments208        for attachment in attachments:209            self.download(attachment)210 211    def _load_documents(self) -> List[Document]:212        """Load all documents in the folder.213 214        Returns:215            List of documents.216        """217        # Create the document loader218        loader = DirectoryLoader(219            path=self.folder_path,220            glob="*.pdf",221            loader_cls=PyPDFLoader,  # type: ignore[arg-type]222        )223        # Load the documents224        documents = loader.load()225        # Return all documents226        return documents227 228    def _get_paths(self, soup: Any) -> List[str]:229        """Get all relative paths in the navbar."""230        relative_paths = []231        course_menu = soup.find("ul", {"class": "courseMenu"})232        if course_menu is None:233            raise ValueError("No course menu found.")234        for link in course_menu.find_all("a"):235            href = link.get("href")236            if href is not None and href.startswith("/"):237                relative_paths.append(href)238        return relative_paths239 240    def download(self, path: str) -> None:241        """Download a file from an url.242 243        Args:244            path: Path to the file.245        """246        # Get the file content247        response = self.session.get(self.base_url + path, allow_redirects=True)248        # Get the filename249        filename = self.parse_filename(response.url)250        # Write the file to disk251        with open(Path(self.folder_path) / filename, "wb") as f:252            f.write(response.content)253 254    def parse_filename(self, url: str) -> str:255        """Parse the filename from an url.256 257        Args:258            url: Url to parse the filename from.259 260        Returns:261            The filename.262        """263        if (url_path := Path(url)) and url_path.suffix == ".pdf":264            return url_path.name265        else:266            return self._parse_filename_from_url(url)267 268    def _parse_filename_from_url(self, url: str) -> str:269        """Parse the filename from an url.270 271        Args:272            url: Url to parse the filename from.273 274        Returns:275            The filename.276 277        Raises:278            ValueError: If the filename could not be parsed.279        """280        filename_matches = re.search(r"filename%2A%3DUTF-8%27%27(.+)", url)281        if filename_matches:282            filename = filename_matches.group(1)283        else:284            raise ValueError(f"Could not parse filename from {url}")285        if ".pdf" not in filename:286            raise ValueError(f"Incorrect file type: {filename}")287        filename = filename.split(".pdf")[0] + ".pdf"288        filename = unquote(filename)289        filename = filename.replace("%20", " ")290        return filename291 292 293if __name__ == "__main__":294    loader = BlackboardLoader(295        "https://<YOUR BLACKBOARD URL"296        " HERE>/webapps/blackboard/content/listContent.jsp?course_id=_<YOUR COURSE ID"297        " HERE>_1&content_id=_<YOUR CONTENT ID HERE>_1&mode=reset",298        "<YOUR BBROUTER COOKIE HERE>",299        load_all_recursively=True,300    )301    documents = loader.load()302    print(f"Loaded {len(documents)} pages of PDFs from {loader.web_path}")  # noqa: T201303 
codekingpro/portable-devtools · Team Ai