Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
chromium.py107 linesDownload Raw Back to document_loaders
1import asyncio2import logging3from typing import AsyncIterator, Iterator, List, Optional4 5from langchain_core.documents import Document6 7from langchain_community.document_loaders.base import BaseLoader8from langchain_community.utils.user_agent import get_user_agent9 10logger = logging.getLogger(__name__)11 12 13class AsyncChromiumLoader(BaseLoader):14    """Scrape HTML pages from URLs using a15    headless instance of the Chromium."""16 17    def __init__(18        self,19        urls: List[str],20        *,21        headless: bool = True,22        user_agent: Optional[str] = None,23    ):24        """Initialize the loader with a list of URL paths.25 26        Args:27            urls: A list of URLs to scrape content from.28            headless: Whether to run browser in headless mode.29            user_agent: The user agent to use for the browser30 31        Raises:32            ImportError: If the required 'playwright' package is not installed.33        """34        self.urls = urls35        self.headless = headless36        self.user_agent = user_agent or get_user_agent()37 38        try:39            import playwright  # noqa: F40140        except ImportError:41            raise ImportError(42                "playwright is required for AsyncChromiumLoader. "43                "Please install it with `pip install playwright`."44            )45 46    async def ascrape_playwright(self, url: str) -> str:47        """48        Asynchronously scrape the content of a given URL using Playwright's async API.49 50        Args:51            url (str): The URL to scrape.52 53        Returns:54            str: The scraped HTML content or an error message if an exception occurs.55 56        """57        from playwright.async_api import async_playwright58 59        logger.info("Starting scraping...")60        results = ""61        async with async_playwright() as p:62            browser = await p.chromium.launch(headless=self.headless)63            try:64                page = await browser.new_page(user_agent=self.user_agent)65                await page.goto(url)66                results = await page.content()  # Simply get the HTML content67                logger.info("Content scraped")68            except Exception as e:69                results = f"Error: {e}"70            await browser.close()71        return results72 73    def lazy_load(self) -> Iterator[Document]:74        """75        Lazily load text content from the provided URLs.76 77        This method yields Documents one at a time as they're scraped,78        instead of waiting to scrape all URLs before returning.79 80        Yields:81            Document: The scraped content encapsulated within a Document object.82 83        """84        for url in self.urls:85            html_content = asyncio.run(self.ascrape_playwright(url))86            metadata = {"source": url}87            yield Document(page_content=html_content, metadata=metadata)88 89    async def alazy_load(self) -> AsyncIterator[Document]:90        """91        Asynchronously load text content from the provided URLs.92 93        This method leverages asyncio to initiate the scraping of all provided URLs94        simultaneously. It improves performance by utilizing concurrent asynchronous95        requests. Each Document is yielded as soon as its content is available,96        encapsulating the scraped content.97 98        Yields:99            Document: A Document object containing the scraped content, along with its100            source URL as metadata.101        """102        tasks = [self.ascrape_playwright(url) for url in self.urls]103        results = await asyncio.gather(*tasks)104        for url, content in zip(self.urls, results):105            metadata = {"source": url}106            yield Document(page_content=content, metadata=metadata)107 
codekingpro/portable-devtools · Team Ai