Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
extract_processor.py190 linesDownload Raw Back to extractor
1import re2import tempfile3from pathlib import Path4from typing import Optional, Union5from urllib.parse import unquote6 7from configs import dify_config8from core.helper import ssrf_proxy9from core.rag.extractor.csv_extractor import CSVExtractor10from core.rag.extractor.entity.datasource_type import DatasourceType11from core.rag.extractor.entity.extract_setting import ExtractSetting12from core.rag.extractor.excel_extractor import ExcelExtractor13from core.rag.extractor.firecrawl.firecrawl_web_extractor import FirecrawlWebExtractor14from core.rag.extractor.html_extractor import HtmlExtractor15from core.rag.extractor.jina_reader_extractor import JinaReaderWebExtractor16from core.rag.extractor.markdown_extractor import MarkdownExtractor17from core.rag.extractor.notion_extractor import NotionExtractor18from core.rag.extractor.pdf_extractor import PdfExtractor19from core.rag.extractor.text_extractor import TextExtractor20from core.rag.extractor.unstructured.unstructured_eml_extractor import UnstructuredEmailExtractor21from core.rag.extractor.unstructured.unstructured_epub_extractor import UnstructuredEpubExtractor22from core.rag.extractor.unstructured.unstructured_markdown_extractor import UnstructuredMarkdownExtractor23from core.rag.extractor.unstructured.unstructured_msg_extractor import UnstructuredMsgExtractor24from core.rag.extractor.unstructured.unstructured_ppt_extractor import UnstructuredPPTExtractor25from core.rag.extractor.unstructured.unstructured_pptx_extractor import UnstructuredPPTXExtractor26from core.rag.extractor.unstructured.unstructured_text_extractor import UnstructuredTextExtractor27from core.rag.extractor.unstructured.unstructured_xml_extractor import UnstructuredXmlExtractor28from core.rag.extractor.word_extractor import WordExtractor29from core.rag.models.document import Document30from extensions.ext_storage import storage31from models.model import UploadFile32 33SUPPORT_URL_CONTENT_TYPES = ["application/pdf", "text/plain", "application/json"]34USER_AGENT = (35    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124"36    " Safari/537.36"37)38 39 40class ExtractProcessor:41    @classmethod42    def load_from_upload_file(43        cls, upload_file: UploadFile, return_text: bool = False, is_automatic: bool = False44    ) -> Union[list[Document], str]:45        extract_setting = ExtractSetting(46            datasource_type="upload_file", upload_file=upload_file, document_model="text_model"47        )48        if return_text:49            delimiter = "\n"50            return delimiter.join([document.page_content for document in cls.extract(extract_setting, is_automatic)])51        else:52            return cls.extract(extract_setting, is_automatic)53 54    @classmethod55    def load_from_url(cls, url: str, return_text: bool = False) -> Union[list[Document], str]:56        response = ssrf_proxy.get(url, headers={"User-Agent": USER_AGENT})57 58        with tempfile.TemporaryDirectory() as temp_dir:59            suffix = Path(url).suffix60            if not suffix and suffix != ".":61                # get content-type62                if response.headers.get("Content-Type"):63                    suffix = "." + response.headers.get("Content-Type").split("/")[-1]64                else:65                    content_disposition = response.headers.get("Content-Disposition")66                    filename_match = re.search(r'filename="([^"]+)"', content_disposition)67                    if filename_match:68                        filename = unquote(filename_match.group(1))69                        suffix = "." + re.search(r"\.(\w+)$", filename).group(1)70 71            file_path = f"{temp_dir}/{next(tempfile._get_candidate_names())}{suffix}"72            Path(file_path).write_bytes(response.content)73            extract_setting = ExtractSetting(datasource_type="upload_file", document_model="text_model")74            if return_text:75                delimiter = "\n"76                return delimiter.join(77                    [78                        document.page_content79                        for document in cls.extract(extract_setting=extract_setting, file_path=file_path)80                    ]81                )82            else:83                return cls.extract(extract_setting=extract_setting, file_path=file_path)84 85    @classmethod86    def extract(87        cls, extract_setting: ExtractSetting, is_automatic: bool = False, file_path: Optional[str] = None88    ) -> list[Document]:89        if extract_setting.datasource_type == DatasourceType.FILE.value:90            with tempfile.TemporaryDirectory() as temp_dir:91                if not file_path:92                    upload_file: UploadFile = extract_setting.upload_file93                    suffix = Path(upload_file.key).suffix94                    file_path = f"{temp_dir}/{next(tempfile._get_candidate_names())}{suffix}"95                    storage.download(upload_file.key, file_path)96                input_file = Path(file_path)97                file_extension = input_file.suffix.lower()98                etl_type = dify_config.ETL_TYPE99                unstructured_api_url = dify_config.UNSTRUCTURED_API_URL100                unstructured_api_key = dify_config.UNSTRUCTURED_API_KEY101                if etl_type == "Unstructured":102                    if file_extension in {".xlsx", ".xls"}:103                        extractor = ExcelExtractor(file_path)104                    elif file_extension == ".pdf":105                        extractor = PdfExtractor(file_path)106                    elif file_extension in {".md", ".markdown"}:107                        extractor = (108                            UnstructuredMarkdownExtractor(file_path, unstructured_api_url, unstructured_api_key)109                            if is_automatic110                            else MarkdownExtractor(file_path, autodetect_encoding=True)111                        )112                    elif file_extension in {".htm", ".html"}:113                        extractor = HtmlExtractor(file_path)114                    elif file_extension == ".docx":115                        extractor = WordExtractor(file_path, upload_file.tenant_id, upload_file.created_by)116                    elif file_extension == ".csv":117                        extractor = CSVExtractor(file_path, autodetect_encoding=True)118                    elif file_extension == ".msg":119                        extractor = UnstructuredMsgExtractor(file_path, unstructured_api_url, unstructured_api_key)120                    elif file_extension == ".eml":121                        extractor = UnstructuredEmailExtractor(file_path, unstructured_api_url, unstructured_api_key)122                    elif file_extension == ".ppt":123                        extractor = UnstructuredPPTExtractor(file_path, unstructured_api_url, unstructured_api_key)124                        # You must first specify the API key125                        # because unstructured_api_key is necessary to parse .ppt documents126                    elif file_extension == ".pptx":127                        extractor = UnstructuredPPTXExtractor(file_path, unstructured_api_url, unstructured_api_key)128                    elif file_extension == ".xml":129                        extractor = UnstructuredXmlExtractor(file_path, unstructured_api_url, unstructured_api_key)130                    elif file_extension == ".epub":131                        extractor = UnstructuredEpubExtractor(file_path, unstructured_api_url, unstructured_api_key)132                    else:133                        # txt134                        extractor = (135                            UnstructuredTextExtractor(file_path, unstructured_api_url)136                            if is_automatic137                            else TextExtractor(file_path, autodetect_encoding=True)138                        )139                else:140                    if file_extension in {".xlsx", ".xls"}:141                        extractor = ExcelExtractor(file_path)142                    elif file_extension == ".pdf":143                        extractor = PdfExtractor(file_path)144                    elif file_extension in {".md", ".markdown"}:145                        extractor = MarkdownExtractor(file_path, autodetect_encoding=True)146                    elif file_extension in {".htm", ".html"}:147                        extractor = HtmlExtractor(file_path)148                    elif file_extension == ".docx":149                        extractor = WordExtractor(file_path, upload_file.tenant_id, upload_file.created_by)150                    elif file_extension == ".csv":151                        extractor = CSVExtractor(file_path, autodetect_encoding=True)152                    elif file_extension == ".epub":153                        extractor = UnstructuredEpubExtractor(file_path)154                    else:155                        # txt156                        extractor = TextExtractor(file_path, autodetect_encoding=True)157                return extractor.extract()158        elif extract_setting.datasource_type == DatasourceType.NOTION.value:159            extractor = NotionExtractor(160                notion_workspace_id=extract_setting.notion_info.notion_workspace_id,161                notion_obj_id=extract_setting.notion_info.notion_obj_id,162                notion_page_type=extract_setting.notion_info.notion_page_type,163                document_model=extract_setting.notion_info.document,164                tenant_id=extract_setting.notion_info.tenant_id,165            )166            return extractor.extract()167        elif extract_setting.datasource_type == DatasourceType.WEBSITE.value:168            if extract_setting.website_info.provider == "firecrawl":169                extractor = FirecrawlWebExtractor(170                    url=extract_setting.website_info.url,171                    job_id=extract_setting.website_info.job_id,172                    tenant_id=extract_setting.website_info.tenant_id,173                    mode=extract_setting.website_info.mode,174                    only_main_content=extract_setting.website_info.only_main_content,175                )176                return extractor.extract()177            elif extract_setting.website_info.provider == "jinareader":178                extractor = JinaReaderWebExtractor(179                    url=extract_setting.website_info.url,180                    job_id=extract_setting.website_info.job_id,181                    tenant_id=extract_setting.website_info.tenant_id,182                    mode=extract_setting.website_info.mode,183                    only_main_content=extract_setting.website_info.only_main_content,184                )185                return extractor.extract()186            else:187                raise ValueError(f"Unsupported website provider: {extract_setting.website_info.provider}")188        else:189            raise ValueError(f"Unsupported datasource type: {extract_setting.datasource_type}")190