Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
firecrawl.py410 linesDownload Raw Back to document_loaders
1import dataclasses2import os3from typing import Any, Iterator, Literal, Optional4 5from langchain_core.document_loaders import BaseLoader6from langchain_core.documents import Document7from langchain_core.utils import get_from_env8 9 10class FireCrawlLoader(BaseLoader):11    """12    FireCrawlLoader document loader integration13 14    Setup:15        Install ``firecrawl-py``,``langchain_community`` and set environment variable ``FIRECRAWL_API_KEY``.16 17        .. code-block:: bash18 19            pip install -U firecrawl-py langchain_community20            export FIRECRAWL_API_KEY="your-api-key"21 22    Instantiate:23        .. code-block:: python24 25            from langchain_community.document_loaders import FireCrawlLoader26 27            loader = FireCrawlLoader(28                url = "https://firecrawl.dev",29                mode = "crawl"30                # other params = ...31            )32 33    Lazy load:34        .. code-block:: python35 36            docs = []37            docs_lazy = loader.lazy_load()38 39            # async variant:40            # docs_lazy = await loader.alazy_load()41 42            for doc in docs_lazy:43                docs.append(doc)44            print(docs[0].page_content[:100])45            print(docs[0].metadata)46 47        .. code-block:: python48 49            Introducing [Smart Crawl!](https://www.firecrawl.dev/smart-crawl)50             Join the waitlist to turn any web51            {'ogUrl': 'https://www.firecrawl.dev/', 'title': 'Home - Firecrawl', 'robots': 'follow, index', 'ogImage': 'https://www.firecrawl.dev/og.png?123', 'ogTitle': 'Firecrawl', 'sitemap': {'lastmod': '2024-08-12T00:28:16.681Z', 'changefreq': 'weekly'}, 'keywords': 'Firecrawl,Markdown,Data,Mendable,Langchain', 'sourceURL': 'https://www.firecrawl.dev/', 'ogSiteName': 'Firecrawl', 'description': 'Firecrawl crawls and converts any website into clean markdown.', 'ogDescription': 'Turn any website into LLM-ready data.', 'pageStatusCode': 200, 'ogLocaleAlternate': []}52 53    Async load:54        .. code-block:: python55 56            docs = await loader.aload()57            print(docs[0].page_content[:100])58            print(docs[0].metadata)59 60        .. code-block:: python61 62            Introducing [Smart Crawl!](https://www.firecrawl.dev/smart-crawl)63             Join the waitlist to turn any web64            {'ogUrl': 'https://www.firecrawl.dev/', 'title': 'Home - Firecrawl', 'robots': 'follow, index', 'ogImage': 'https://www.firecrawl.dev/og.png?123', 'ogTitle': 'Firecrawl', 'sitemap': {'lastmod': '2024-08-12T00:28:16.681Z', 'changefreq': 'weekly'}, 'keywords': 'Firecrawl,Markdown,Data,Mendable,Langchain', 'sourceURL': 'https://www.firecrawl.dev/', 'ogSiteName': 'Firecrawl', 'description': 'Firecrawl crawls and converts any website into clean markdown.', 'ogDescription': 'Turn any website into LLM-ready data.', 'pageStatusCode': 200, 'ogLocaleAlternate': []}65 66    """  # noqa: E50167 68    # No legacy support in v2-only implementation69 70    def __init__(71        self,72        url: Optional[str] = None,73        *,74        query: Optional[str] = None,75        api_key: Optional[str] = None,76        api_url: Optional[str] = None,77        mode: Literal["crawl", "scrape", "map", "extract", "search"] = "crawl",78        params: Optional[dict] = None,79    ):80        """Initialize with API key and url.81 82        Args:83            url: The url to be crawled.84            api_key: The Firecrawl API key. If not specified will be read from env var85                FIRECRAWL_API_KEY. Get an API key86            api_url: The Firecrawl API URL. If not specified will be read from env var87                FIRECRAWL_API_URL or defaults to https://api.firecrawl.dev.88            mode: The mode to run the loader in. Default is "crawl".89                 Options include "scrape" (single url),90                 "crawl" (all accessible sub pages),91                 "map" (returns list of links that are semantically related).92                 "extract" (extracts structured data from a page).93                 "search" (search for data across the web).94            params: The parameters to pass to the Firecrawl API.95                Examples include crawlerOptions.96                For more details, visit: https://github.com/mendableai/firecrawl-py97        """98 99        try:100            from firecrawl import FirecrawlApp101        except ImportError:102            raise ImportError(103                "`firecrawl` package not found, please run `pip install firecrawl-py`"104            )105        if mode not in ("crawl", "scrape", "search", "map", "extract", "search"):106            raise ValueError(107                f"""Invalid mode '{mode}'.108                Allowed: 'crawl', 'scrape', 'search', 'map', 'extract', 'search'."""109            )110 111        if mode in ("scrape", "crawl", "map", "extract") and not url:112            raise ValueError("Url must be provided for modes other than 'search'")113        if mode == "search" and not (query or (params and params.get("query"))):114            raise ValueError("Query must be provided for search mode")115 116        api_key = api_key or get_from_env("api_key", "FIRECRAWL_API_KEY")117        # Ensure we never pass None for api_url (v2 client validates as str).118        # Avoid get_from_env to prevent raising.119        resolved_api_url = (120            api_url or os.getenv("FIRECRAWL_API_URL") or "https://api.firecrawl.dev"121        )122        self.firecrawl = FirecrawlApp(api_key=api_key, api_url=resolved_api_url)123        self.url = url or ""124        self.mode = mode125        self.params = params or {}126        if query is not None:127            self.params["query"] = query128 129    def lazy_load(self) -> Iterator[Document]:130        # Prepare integration tag and filter params per method131        firecrawl_docs: list[Any] = []132        if self.mode == "scrape":133            allowed = {134                "formats",135                "headers",136                "include_tags",137                "exclude_tags",138                "only_main_content",139                "timeout",140                "wait_for",141                "mobile",142                "parsers",143                "actions",144                "location",145                "skip_tls_verification",146                "remove_base64_images",147                "fast_mode",148                "use_mock",149                "block_ads",150                "proxy",151                "max_age",152                "store_in_cache",153            }154            kwargs = {k: v for k, v in self.params.items() if k in allowed}155            kwargs["integration"] = "langchain"156            firecrawl_docs = [self.firecrawl.scrape(self.url, **kwargs)]157        elif self.mode == "crawl":158            if not self.url:159                raise ValueError("URL is required for crawl mode")160            allowed = {161                "prompt",162                "exclude_paths",163                "include_paths",164                "max_discovery_depth",165                "ignore_sitemap",166                "ignore_query_parameters",167                "limit",168                "crawl_entire_domain",169                "allow_external_links",170                "allow_subdomains",171                "delay",172                "max_concurrency",173                "webhook",174                "scrape_options",175                "zero_data_retention",176                "poll_interval",177                "timeout",178            }179            kwargs = {k: v for k, v in self.params.items() if k in allowed}180            kwargs["integration"] = "langchain"181            crawl_response = self.firecrawl.crawl(self.url, **kwargs)182            # Support dict or object with 'data'183            if isinstance(crawl_response, dict):184                data = crawl_response.get("data", [])185                firecrawl_docs = list(data) if isinstance(data, list) else []186            else:187                data = getattr(crawl_response, "data", [])188                firecrawl_docs = list(data) if isinstance(data, list) else []189        elif self.mode == "map":190            if not self.url:191                raise ValueError("URL is required for map mode")192            allowed = {193                "search",194                "include_subdomains",195                "limit",196                "sitemap",197                "timeout",198                "location",199            }200            kwargs = {k: v for k, v in self.params.items() if k in allowed}201            kwargs["integration"] = "langchain"202            map_response = self.firecrawl.map(self.url, **kwargs)203            # Firecrawl v2 (>=4.3.6) returns an object with a `links` array204            # Fallback to legacy list response if needed205            if isinstance(map_response, dict):206                links = map_response.get("links")207                firecrawl_docs = list(links) if isinstance(links, list) else []208            elif hasattr(map_response, "links"):209                links = getattr(map_response, "links")210                firecrawl_docs = list(links) if isinstance(links, list) else []211            else:212                is_list = isinstance(map_response, list)213                firecrawl_docs = list(map_response) if is_list else []214        elif self.mode == "extract":215            if not self.url:216                raise ValueError("URL is required for extract mode")217            allowed = {218                "prompt",219                "schema",220                "system_prompt",221                "allow_external_links",222                "enable_web_search",223                "show_sources",224                "scrape_options",225                "ignore_invalid_urls",226                "poll_interval",227                "timeout",228                "agent",229            }230            kwargs = {k: v for k, v in self.params.items() if k in allowed}231            kwargs["integration"] = "langchain"232            firecrawl_docs = [str(self.firecrawl.extract([self.url], **kwargs))]233        elif self.mode == "search":234            allowed = {235                "sources",236                "categories",237                "limit",238                "tbs",239                "location",240                "ignore_invalid_urls",241                "timeout",242                "scrape_options",243            }244            kwargs = {k: v for k, v in self.params.items() if k in allowed}245            kwargs["integration"] = "langchain"246            search_data = self.firecrawl.search(247                query=self.params.get("query"), **kwargs248            )249            # If SDK already returns a list[dict], use it directly250            if isinstance(search_data, list):251                firecrawl_docs = list(search_data)252            else:253                # Normalize typed SearchData into list of dicts with markdown + metadata254                results: list[dict[str, Any]] = []255                containers = []256                if isinstance(search_data, dict):257                    containers = [258                        search_data.get("web"),259                        search_data.get("news"),260                        search_data.get("images"),261                    ]262                else:263                    containers = [264                        getattr(search_data, "web", None),265                        getattr(search_data, "news", None),266                        getattr(search_data, "images", None),267                    ]268                for kind, items in (269                    ("web", containers[0]),270                    ("news", containers[1]),271                    ("images", containers[2]),272                ):273                    if not items:274                        continue275                    for item in items:276                        url_val = (277                            getattr(item, "url", None)278                            if not isinstance(item, dict)279                            else item.get("url")280                        )281                        title_val = (282                            getattr(item, "title", None)283                            if not isinstance(item, dict)284                            else item.get("title")285                        )286                        desc_val = (287                            getattr(item, "description", None)288                            if not isinstance(item, dict)289                            else item.get("description")290                        )291                        content_val = desc_val or title_val or url_val or ""292                        metadata_val = {293                            k: v294                            for k, v in {295                                "url": url_val,296                                "title": title_val,297                                "category": getattr(item, "category", None)298                                if not isinstance(item, dict)299                                else item.get("category"),300                                "type": kind,301                            }.items()302                            if v is not None303                        }304                        results.append(305                            {"markdown": content_val, "metadata": metadata_val}306                        )307                firecrawl_docs = results308        else:309            raise ValueError(310                f"""Invalid mode '{self.mode}'.311                Allowed: 'crawl', 'scrape', 'map', 'extract', 'search'."""312            )313        for doc in firecrawl_docs:314            if self.mode == "map":315                # Support both legacy string list and v2 link objects316                if isinstance(doc, str):317                    page_content: str = doc318                    meta: dict[str, Any] = {}319                elif isinstance(doc, dict):320                    page_content_value = doc.get("url") or doc.get("href") or ""321                    page_content = (322                        page_content_value323                        if isinstance(page_content_value, str)324                        else str(page_content_value or "")325                    )326                    meta = {327                        k: v328                        for k, v in {329                            "title": doc.get("title"),330                            "description": doc.get("description"),331                        }.items()332                        if v is not None333                    }334                elif hasattr(doc, "url") or hasattr(doc, "title"):335                    page_content_value = getattr(doc, "url", "") or getattr(336                        doc, "href", ""337                    )338                    page_content = (339                        page_content_value340                        if isinstance(page_content_value, str)341                        else str(page_content_value or "")342                    )343                    meta = {}344                    title = getattr(doc, "title", None)345                    description = getattr(doc, "description", None)346                    if title is not None:347                        meta["title"] = title348                    if description is not None:349                        meta["description"] = description350                else:351                    page_content = str(doc)352                    meta = {}353            elif self.mode == "extract":354                page_content = str(doc)355                meta = {}356            elif self.mode == "search":357                # Already normalized to dicts with markdown/metadata above358                if isinstance(doc, dict):359                    markdown_value = doc.get("markdown") or ""360                    page_content = (361                        markdown_value362                        if isinstance(markdown_value, str)363                        else str(markdown_value or "")364                    )365                    metadata_obj = doc.get("metadata", {})366                    meta = metadata_obj if isinstance(metadata_obj, dict) else {}367                else:368                    page_content = str(doc)369                    meta = {}370            else:371                if isinstance(doc, dict):372                    content_value = (373                        doc.get("markdown") or doc.get("html") or doc.get("rawHtml", "")374                    )375                    page_content = (376                        content_value377                        if isinstance(content_value, str)378                        else str(content_value or "")379                    )380                    meta = doc.get("metadata", {})381                else:382                    content_value = (383                        getattr(doc, "markdown", None)384                        or getattr(doc, "html", None)385                        or getattr(doc, "rawHtml", "")386                    )387                    page_content = (388                        content_value389                        if isinstance(content_value, str)390                        else str(content_value or "")391                    )392                    meta = getattr(doc, "metadata", {}) or {}393 394                # Normalize metadata to plain dict for LangChain Document395                if not isinstance(meta, dict):396                    if hasattr(meta, "model_dump") and callable(meta.model_dump):397                        meta = meta.model_dump()398                    elif dataclasses.is_dataclass(meta):399                        meta = dataclasses.asdict(meta)  # type: ignore[arg-type]400                    elif hasattr(meta, "__dict__"):401                        meta = dict(vars(meta))402                    else:403                        meta = {"value": str(meta)}404            if not page_content:405                continue406            yield Document(407                page_content=page_content,408                metadata=meta,409            )410 
codekingpro/portable-devtools · Team Ai