codekingpro/portable-devtools
114k
1import contextlib2import re3from pathlib import Path4from typing import Any, List, Optional, Tuple5from urllib.parse import unquote6 7from langchain_core.documents import Document8 9from langchain_community.document_loaders.directory import DirectoryLoader10from langchain_community.document_loaders.pdf import PyPDFLoader11from langchain_community.document_loaders.web_base import WebBaseLoader12 13 14class BlackboardLoader(WebBaseLoader):15 """Load a `Blackboard` course.16 17 This loader is not compatible with all Blackboard courses. It is only18 compatible with courses that use the new Blackboard interface.19 To use this loader, you must have the BbRouter cookie. You can get this20 cookie by logging into the course and then copying the value of the21 BbRouter cookie from the browser's developer tools.22 23 Example:24 .. code-block:: python25 26 from langchain_community.document_loaders import BlackboardLoader27 28 loader = BlackboardLoader(29 blackboard_course_url="https://blackboard.example.com/webapps/blackboard/execute/announcement?method=search&context=course_entry&course_id=_123456_1",30 bbrouter="expires:12345...",31 )32 documents = loader.load()33 34 """35 36 def __init__(37 self,38 blackboard_course_url: str,39 bbrouter: str,40 load_all_recursively: bool = True,41 basic_auth: Optional[Tuple[str, str]] = None,42 cookies: Optional[dict] = None,43 continue_on_failure: bool = False,44 show_progress: bool = True,45 ):46 """Initialize with blackboard course url.47 48 The BbRouter cookie is required for most blackboard courses.49 50 Args:51 blackboard_course_url: Blackboard course url.52 bbrouter: BbRouter cookie.53 load_all_recursively: If True, load all documents recursively.54 basic_auth: Basic auth credentials.55 cookies: Cookies.56 continue_on_failure: whether to continue loading the sitemap if an error57 occurs loading a url, emitting a warning instead of raising an58 exception. Setting this to True makes the loader more robust, but also59 may result in missing data. Default: False60 show_progress: whether to show a progress bar while loading. Default: True61 62 Raises:63 ValueError: If blackboard course url is invalid.64 """65 super().__init__(66 web_paths=(blackboard_course_url),67 continue_on_failure=continue_on_failure,68 show_progress=show_progress,69 )70 # Get base url71 try:72 self.base_url = blackboard_course_url.split("/webapps/blackboard")[0]73 except IndexError:74 raise IndexError(75 "Invalid blackboard course url. "76 "Please provide a url that starts with "77 "https://<blackboard_url>/webapps/blackboard"78 )79 if basic_auth is not None:80 self.session.auth = basic_auth81 # Combine cookies82 if cookies is None:83 cookies = {}84 cookies.update({"BbRouter": bbrouter})85 self.session.cookies.update(cookies)86 self.load_all_recursively = load_all_recursively87 self.check_bs4()88 89 def check_bs4(self) -> None:90 """Check if BeautifulSoup4 is installed.91 92 Raises:93 ImportError: If BeautifulSoup4 is not installed.94 """95 try:96 import bs4 # noqa: F40197 except ImportError:98 raise ImportError(99 "BeautifulSoup4 is required for BlackboardLoader. "100 "Please install it with `pip install beautifulsoup4`."101 )102 103 def load(self) -> List[Document]:104 """Load data into Document objects.105 106 Returns:107 List of Documents.108 """109 if self.load_all_recursively:110 soup_info = self.scrape()111 self.folder_path = self._get_folder_path(soup_info)112 relative_paths = self._get_paths(soup_info)113 documents = []114 for path in relative_paths:115 url = self.base_url + path116 print(f"Fetching documents from {url}") # noqa: T201117 soup_info = self._scrape(url)118 with contextlib.suppress(ValueError):119 documents.extend(self._get_documents(soup_info))120 return documents121 else:122 print(f"Fetching documents from {self.web_path}") # noqa: T201123 soup_info = self.scrape()124 self.folder_path = self._get_folder_path(soup_info)125 return self._get_documents(soup_info)126 127 def _get_folder_path(self, soup: Any) -> str:128 """Get the folder path to save the Documents in.129 130 Args:131 soup: BeautifulSoup4 soup object.132 133 Returns:134 Folder path.135 """136 # Get the course name137 course_name = soup.find("span", {"id": "crumb_1"})138 if course_name is None:139 raise ValueError("No course name found.")140 course_name = course_name.text.strip()141 # Prepare the folder path142 course_name_clean = (143 unquote(course_name)144 .replace(" ", "_")145 .replace("/", "_")146 .replace(":", "_")147 .replace(",", "_")148 .replace("?", "_")149 .replace("'", "_")150 .replace("!", "_")151 .replace('"', "_")152 )153 # Get the folder path154 folder_path = Path(".") / course_name_clean155 return str(folder_path)156 157 def _get_documents(self, soup: Any) -> List[Document]:158 """Fetch content from page and return Documents.159 160 Args:161 soup: BeautifulSoup4 soup object.162 163 Returns:164 List of documents.165 """166 attachments = self._get_attachments(soup)167 self._download_attachments(attachments)168 documents = self._load_documents()169 return documents170 171 def _get_attachments(self, soup: Any) -> List[str]:172 """Get all attachments from a page.173 174 Args:175 soup: BeautifulSoup4 soup object.176 177 Returns:178 List of attachments.179 """180 from bs4 import BeautifulSoup, Tag181 182 # Get content list183 content_list: BeautifulSoup184 content_list = soup.find("ul", {"class": "contentList"})185 if content_list is None:186 raise ValueError("No content list found.")187 # Get all attachments188 attachments = []189 attachment: Tag190 for attachment in content_list.find_all("ul", {"class": "attachments"}):191 link: Tag192 for link in attachment.find_all("a"):193 href = link.get("href")194 # Only add if href is not None and does not start with #195 if href is not None and not href.startswith("#"): # type: ignore[union-attr]196 attachments.append(href)197 return attachments # type: ignore[return-value]198 199 def _download_attachments(self, attachments: List[str]) -> None:200 """Download all attachments.201 202 Args:203 attachments: List of attachments.204 """205 # Make sure the folder exists206 Path(self.folder_path).mkdir(parents=True, exist_ok=True)207 # Download all attachments208 for attachment in attachments:209 self.download(attachment)210 211 def _load_documents(self) -> List[Document]:212 """Load all documents in the folder.213 214 Returns:215 List of documents.216 """217 # Create the document loader218 loader = DirectoryLoader(219 path=self.folder_path,220 glob="*.pdf",221 loader_cls=PyPDFLoader, # type: ignore[arg-type]222 )223 # Load the documents224 documents = loader.load()225 # Return all documents226 return documents227 228 def _get_paths(self, soup: Any) -> List[str]:229 """Get all relative paths in the navbar."""230 relative_paths = []231 course_menu = soup.find("ul", {"class": "courseMenu"})232 if course_menu is None:233 raise ValueError("No course menu found.")234 for link in course_menu.find_all("a"):235 href = link.get("href")236 if href is not None and href.startswith("/"):237 relative_paths.append(href)238 return relative_paths239 240 def download(self, path: str) -> None:241 """Download a file from an url.242 243 Args:244 path: Path to the file.245 """246 # Get the file content247 response = self.session.get(self.base_url + path, allow_redirects=True)248 # Get the filename249 filename = self.parse_filename(response.url)250 # Write the file to disk251 with open(Path(self.folder_path) / filename, "wb") as f:252 f.write(response.content)253 254 def parse_filename(self, url: str) -> str:255 """Parse the filename from an url.256 257 Args:258 url: Url to parse the filename from.259 260 Returns:261 The filename.262 """263 if (url_path := Path(url)) and url_path.suffix == ".pdf":264 return url_path.name265 else:266 return self._parse_filename_from_url(url)267 268 def _parse_filename_from_url(self, url: str) -> str:269 """Parse the filename from an url.270 271 Args:272 url: Url to parse the filename from.273 274 Returns:275 The filename.276 277 Raises:278 ValueError: If the filename could not be parsed.279 """280 filename_matches = re.search(r"filename%2A%3DUTF-8%27%27(.+)", url)281 if filename_matches:282 filename = filename_matches.group(1)283 else:284 raise ValueError(f"Could not parse filename from {url}")285 if ".pdf" not in filename:286 raise ValueError(f"Incorrect file type: {filename}")287 filename = filename.split(".pdf")[0] + ".pdf"288 filename = unquote(filename)289 filename = filename.replace("%20", " ")290 return filename291 292 293if __name__ == "__main__":294 loader = BlackboardLoader(295 "https://<YOUR BLACKBOARD URL"296 " HERE>/webapps/blackboard/content/listContent.jsp?course_id=_<YOUR COURSE ID"297 " HERE>_1&content_id=_<YOUR CONTENT ID HERE>_1&mode=reset",298 "<YOUR BBROUTER COOKIE HERE>",299 load_all_recursively=True,300 )301 documents = loader.load()302 print(f"Loaded {len(documents)} pages of PDFs from {loader.web_path}") # noqa: T201303 