Underground-Digital/Workflow-Engine
0
1import re2import tempfile3from pathlib import Path4from typing import Optional, Union5from urllib.parse import unquote6 7from configs import dify_config8from core.helper import ssrf_proxy9from core.rag.extractor.csv_extractor import CSVExtractor10from core.rag.extractor.entity.datasource_type import DatasourceType11from core.rag.extractor.entity.extract_setting import ExtractSetting12from core.rag.extractor.excel_extractor import ExcelExtractor13from core.rag.extractor.firecrawl.firecrawl_web_extractor import FirecrawlWebExtractor14from core.rag.extractor.html_extractor import HtmlExtractor15from core.rag.extractor.jina_reader_extractor import JinaReaderWebExtractor16from core.rag.extractor.markdown_extractor import MarkdownExtractor17from core.rag.extractor.notion_extractor import NotionExtractor18from core.rag.extractor.pdf_extractor import PdfExtractor19from core.rag.extractor.text_extractor import TextExtractor20from core.rag.extractor.unstructured.unstructured_eml_extractor import UnstructuredEmailExtractor21from core.rag.extractor.unstructured.unstructured_epub_extractor import UnstructuredEpubExtractor22from core.rag.extractor.unstructured.unstructured_markdown_extractor import UnstructuredMarkdownExtractor23from core.rag.extractor.unstructured.unstructured_msg_extractor import UnstructuredMsgExtractor24from core.rag.extractor.unstructured.unstructured_ppt_extractor import UnstructuredPPTExtractor25from core.rag.extractor.unstructured.unstructured_pptx_extractor import UnstructuredPPTXExtractor26from core.rag.extractor.unstructured.unstructured_text_extractor import UnstructuredTextExtractor27from core.rag.extractor.unstructured.unstructured_xml_extractor import UnstructuredXmlExtractor28from core.rag.extractor.word_extractor import WordExtractor29from core.rag.models.document import Document30from extensions.ext_storage import storage31from models.model import UploadFile32 33SUPPORT_URL_CONTENT_TYPES = ["application/pdf", "text/plain", "application/json"]34USER_AGENT = (35 "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124"36 " Safari/537.36"37)38 39 40class ExtractProcessor:41 @classmethod42 def load_from_upload_file(43 cls, upload_file: UploadFile, return_text: bool = False, is_automatic: bool = False44 ) -> Union[list[Document], str]:45 extract_setting = ExtractSetting(46 datasource_type="upload_file", upload_file=upload_file, document_model="text_model"47 )48 if return_text:49 delimiter = "\n"50 return delimiter.join([document.page_content for document in cls.extract(extract_setting, is_automatic)])51 else:52 return cls.extract(extract_setting, is_automatic)53 54 @classmethod55 def load_from_url(cls, url: str, return_text: bool = False) -> Union[list[Document], str]:56 response = ssrf_proxy.get(url, headers={"User-Agent": USER_AGENT})57 58 with tempfile.TemporaryDirectory() as temp_dir:59 suffix = Path(url).suffix60 if not suffix and suffix != ".":61 # get content-type62 if response.headers.get("Content-Type"):63 suffix = "." + response.headers.get("Content-Type").split("/")[-1]64 else:65 content_disposition = response.headers.get("Content-Disposition")66 filename_match = re.search(r'filename="([^"]+)"', content_disposition)67 if filename_match:68 filename = unquote(filename_match.group(1))69 suffix = "." + re.search(r"\.(\w+)$", filename).group(1)70 71 file_path = f"{temp_dir}/{next(tempfile._get_candidate_names())}{suffix}"72 Path(file_path).write_bytes(response.content)73 extract_setting = ExtractSetting(datasource_type="upload_file", document_model="text_model")74 if return_text:75 delimiter = "\n"76 return delimiter.join(77 [78 document.page_content79 for document in cls.extract(extract_setting=extract_setting, file_path=file_path)80 ]81 )82 else:83 return cls.extract(extract_setting=extract_setting, file_path=file_path)84 85 @classmethod86 def extract(87 cls, extract_setting: ExtractSetting, is_automatic: bool = False, file_path: Optional[str] = None88 ) -> list[Document]:89 if extract_setting.datasource_type == DatasourceType.FILE.value:90 with tempfile.TemporaryDirectory() as temp_dir:91 if not file_path:92 upload_file: UploadFile = extract_setting.upload_file93 suffix = Path(upload_file.key).suffix94 file_path = f"{temp_dir}/{next(tempfile._get_candidate_names())}{suffix}"95 storage.download(upload_file.key, file_path)96 input_file = Path(file_path)97 file_extension = input_file.suffix.lower()98 etl_type = dify_config.ETL_TYPE99 unstructured_api_url = dify_config.UNSTRUCTURED_API_URL100 unstructured_api_key = dify_config.UNSTRUCTURED_API_KEY101 if etl_type == "Unstructured":102 if file_extension in {".xlsx", ".xls"}:103 extractor = ExcelExtractor(file_path)104 elif file_extension == ".pdf":105 extractor = PdfExtractor(file_path)106 elif file_extension in {".md", ".markdown"}:107 extractor = (108 UnstructuredMarkdownExtractor(file_path, unstructured_api_url, unstructured_api_key)109 if is_automatic110 else MarkdownExtractor(file_path, autodetect_encoding=True)111 )112 elif file_extension in {".htm", ".html"}:113 extractor = HtmlExtractor(file_path)114 elif file_extension == ".docx":115 extractor = WordExtractor(file_path, upload_file.tenant_id, upload_file.created_by)116 elif file_extension == ".csv":117 extractor = CSVExtractor(file_path, autodetect_encoding=True)118 elif file_extension == ".msg":119 extractor = UnstructuredMsgExtractor(file_path, unstructured_api_url, unstructured_api_key)120 elif file_extension == ".eml":121 extractor = UnstructuredEmailExtractor(file_path, unstructured_api_url, unstructured_api_key)122 elif file_extension == ".ppt":123 extractor = UnstructuredPPTExtractor(file_path, unstructured_api_url, unstructured_api_key)124 # You must first specify the API key125 # because unstructured_api_key is necessary to parse .ppt documents126 elif file_extension == ".pptx":127 extractor = UnstructuredPPTXExtractor(file_path, unstructured_api_url, unstructured_api_key)128 elif file_extension == ".xml":129 extractor = UnstructuredXmlExtractor(file_path, unstructured_api_url, unstructured_api_key)130 elif file_extension == ".epub":131 extractor = UnstructuredEpubExtractor(file_path, unstructured_api_url, unstructured_api_key)132 else:133 # txt134 extractor = (135 UnstructuredTextExtractor(file_path, unstructured_api_url)136 if is_automatic137 else TextExtractor(file_path, autodetect_encoding=True)138 )139 else:140 if file_extension in {".xlsx", ".xls"}:141 extractor = ExcelExtractor(file_path)142 elif file_extension == ".pdf":143 extractor = PdfExtractor(file_path)144 elif file_extension in {".md", ".markdown"}:145 extractor = MarkdownExtractor(file_path, autodetect_encoding=True)146 elif file_extension in {".htm", ".html"}:147 extractor = HtmlExtractor(file_path)148 elif file_extension == ".docx":149 extractor = WordExtractor(file_path, upload_file.tenant_id, upload_file.created_by)150 elif file_extension == ".csv":151 extractor = CSVExtractor(file_path, autodetect_encoding=True)152 elif file_extension == ".epub":153 extractor = UnstructuredEpubExtractor(file_path)154 else:155 # txt156 extractor = TextExtractor(file_path, autodetect_encoding=True)157 return extractor.extract()158 elif extract_setting.datasource_type == DatasourceType.NOTION.value:159 extractor = NotionExtractor(160 notion_workspace_id=extract_setting.notion_info.notion_workspace_id,161 notion_obj_id=extract_setting.notion_info.notion_obj_id,162 notion_page_type=extract_setting.notion_info.notion_page_type,163 document_model=extract_setting.notion_info.document,164 tenant_id=extract_setting.notion_info.tenant_id,165 )166 return extractor.extract()167 elif extract_setting.datasource_type == DatasourceType.WEBSITE.value:168 if extract_setting.website_info.provider == "firecrawl":169 extractor = FirecrawlWebExtractor(170 url=extract_setting.website_info.url,171 job_id=extract_setting.website_info.job_id,172 tenant_id=extract_setting.website_info.tenant_id,173 mode=extract_setting.website_info.mode,174 only_main_content=extract_setting.website_info.only_main_content,175 )176 return extractor.extract()177 elif extract_setting.website_info.provider == "jinareader":178 extractor = JinaReaderWebExtractor(179 url=extract_setting.website_info.url,180 job_id=extract_setting.website_info.job_id,181 tenant_id=extract_setting.website_info.tenant_id,182 mode=extract_setting.website_info.mode,183 only_main_content=extract_setting.website_info.only_main_content,184 )185 return extractor.extract()186 else:187 raise ValueError(f"Unsupported website provider: {extract_setting.website_info.provider}")188 else:189 raise ValueError(f"Unsupported datasource type: {extract_setting.datasource_type}")190 