Underground-Digital/Workflow-Engine
0
1import datetime2import json3 4import requests5from flask_login import current_user6 7from core.helper import encrypter8from core.rag.extractor.firecrawl.firecrawl_app import FirecrawlApp9from extensions.ext_redis import redis_client10from extensions.ext_storage import storage11from services.auth.api_key_auth_service import ApiKeyAuthService12 13 14class WebsiteService:15 @classmethod16 def document_create_args_validate(cls, args: dict):17 if "url" not in args or not args["url"]:18 raise ValueError("url is required")19 if "options" not in args or not args["options"]:20 raise ValueError("options is required")21 if "limit" not in args["options"] or not args["options"]["limit"]:22 raise ValueError("limit is required")23 24 @classmethod25 def crawl_url(cls, args: dict) -> dict:26 provider = args.get("provider")27 url = args.get("url")28 options = args.get("options")29 credentials = ApiKeyAuthService.get_auth_credentials(current_user.current_tenant_id, "website", provider)30 if provider == "firecrawl":31 # decrypt api_key32 api_key = encrypter.decrypt_token(33 tenant_id=current_user.current_tenant_id, token=credentials.get("config").get("api_key")34 )35 firecrawl_app = FirecrawlApp(api_key=api_key, base_url=credentials.get("config").get("base_url", None))36 crawl_sub_pages = options.get("crawl_sub_pages", False)37 only_main_content = options.get("only_main_content", False)38 if not crawl_sub_pages:39 params = {40 "crawlerOptions": {41 "includes": [],42 "excludes": [],43 "generateImgAltText": True,44 "limit": 1,45 "returnOnlyUrls": False,46 "pageOptions": {"onlyMainContent": only_main_content, "includeHtml": False},47 }48 }49 else:50 includes = options.get("includes").split(",") if options.get("includes") else []51 excludes = options.get("excludes").split(",") if options.get("excludes") else []52 params = {53 "crawlerOptions": {54 "includes": includes or [],55 "excludes": excludes or [],56 "generateImgAltText": True,57 "limit": options.get("limit", 1),58 "returnOnlyUrls": False,59 "pageOptions": {"onlyMainContent": only_main_content, "includeHtml": False},60 }61 }62 if options.get("max_depth"):63 params["crawlerOptions"]["maxDepth"] = options.get("max_depth")64 job_id = firecrawl_app.crawl_url(url, params)65 website_crawl_time_cache_key = f"website_crawl_{job_id}"66 time = str(datetime.datetime.now().timestamp())67 redis_client.setex(website_crawl_time_cache_key, 3600, time)68 return {"status": "active", "job_id": job_id}69 elif provider == "jinareader":70 api_key = encrypter.decrypt_token(71 tenant_id=current_user.current_tenant_id, token=credentials.get("config").get("api_key")72 )73 crawl_sub_pages = options.get("crawl_sub_pages", False)74 if not crawl_sub_pages:75 response = requests.get(76 f"https://r.jina.ai/{url}",77 headers={"Accept": "application/json", "Authorization": f"Bearer {api_key}"},78 )79 if response.json().get("code") != 200:80 raise ValueError("Failed to crawl")81 return {"status": "active", "data": response.json().get("data")}82 else:83 response = requests.post(84 "https://adaptivecrawl-kir3wx7b3a-uc.a.run.app",85 json={86 "url": url,87 "maxPages": options.get("limit", 1),88 "useSitemap": options.get("use_sitemap", True),89 },90 headers={91 "Content-Type": "application/json",92 "Authorization": f"Bearer {api_key}",93 },94 )95 if response.json().get("code") != 200:96 raise ValueError("Failed to crawl")97 return {"status": "active", "job_id": response.json().get("data", {}).get("taskId")}98 else:99 raise ValueError("Invalid provider")100 101 @classmethod102 def get_crawl_status(cls, job_id: str, provider: str) -> dict:103 credentials = ApiKeyAuthService.get_auth_credentials(current_user.current_tenant_id, "website", provider)104 if provider == "firecrawl":105 # decrypt api_key106 api_key = encrypter.decrypt_token(107 tenant_id=current_user.current_tenant_id, token=credentials.get("config").get("api_key")108 )109 firecrawl_app = FirecrawlApp(api_key=api_key, base_url=credentials.get("config").get("base_url", None))110 result = firecrawl_app.check_crawl_status(job_id)111 crawl_status_data = {112 "status": result.get("status", "active"),113 "job_id": job_id,114 "total": result.get("total", 0),115 "current": result.get("current", 0),116 "data": result.get("data", []),117 }118 if crawl_status_data["status"] == "completed":119 website_crawl_time_cache_key = f"website_crawl_{job_id}"120 start_time = redis_client.get(website_crawl_time_cache_key)121 if start_time:122 end_time = datetime.datetime.now().timestamp()123 time_consuming = abs(end_time - float(start_time))124 crawl_status_data["time_consuming"] = f"{time_consuming:.2f}"125 redis_client.delete(website_crawl_time_cache_key)126 elif provider == "jinareader":127 api_key = encrypter.decrypt_token(128 tenant_id=current_user.current_tenant_id, token=credentials.get("config").get("api_key")129 )130 response = requests.post(131 "https://adaptivecrawlstatus-kir3wx7b3a-uc.a.run.app",132 headers={"Content-Type": "application/json", "Authorization": f"Bearer {api_key}"},133 json={"taskId": job_id},134 )135 data = response.json().get("data", {})136 crawl_status_data = {137 "status": data.get("status", "active"),138 "job_id": job_id,139 "total": len(data.get("urls", [])),140 "current": len(data.get("processed", [])) + len(data.get("failed", [])),141 "data": [],142 "time_consuming": data.get("duration", 0) / 1000,143 }144 145 if crawl_status_data["status"] == "completed":146 response = requests.post(147 "https://adaptivecrawlstatus-kir3wx7b3a-uc.a.run.app",148 headers={"Content-Type": "application/json", "Authorization": f"Bearer {api_key}"},149 json={"taskId": job_id, "urls": list(data.get("processed", {}).keys())},150 )151 data = response.json().get("data", {})152 formatted_data = [153 {154 "title": item.get("data", {}).get("title"),155 "source_url": item.get("data", {}).get("url"),156 "description": item.get("data", {}).get("description"),157 "markdown": item.get("data", {}).get("content"),158 }159 for item in data.get("processed", {}).values()160 ]161 crawl_status_data["data"] = formatted_data162 else:163 raise ValueError("Invalid provider")164 return crawl_status_data165 166 @classmethod167 def get_crawl_url_data(cls, job_id: str, provider: str, url: str, tenant_id: str) -> dict | None:168 credentials = ApiKeyAuthService.get_auth_credentials(tenant_id, "website", provider)169 # decrypt api_key170 api_key = encrypter.decrypt_token(tenant_id=tenant_id, token=credentials.get("config").get("api_key"))171 if provider == "firecrawl":172 file_key = "website_files/" + job_id + ".txt"173 if storage.exists(file_key):174 data = storage.load_once(file_key)175 if data:176 data = json.loads(data.decode("utf-8"))177 else:178 firecrawl_app = FirecrawlApp(api_key=api_key, base_url=credentials.get("config").get("base_url", None))179 result = firecrawl_app.check_crawl_status(job_id)180 if result.get("status") != "completed":181 raise ValueError("Crawl job is not completed")182 data = result.get("data")183 if data:184 for item in data:185 if item.get("source_url") == url:186 return item187 return None188 elif provider == "jinareader":189 file_key = "website_files/" + job_id + ".txt"190 if storage.exists(file_key):191 data = storage.load_once(file_key)192 if data:193 data = json.loads(data.decode("utf-8"))194 elif not job_id:195 response = requests.get(196 f"https://r.jina.ai/{url}",197 headers={"Accept": "application/json", "Authorization": f"Bearer {api_key}"},198 )199 if response.json().get("code") != 200:200 raise ValueError("Failed to crawl")201 return response.json().get("data")202 else:203 api_key = encrypter.decrypt_token(tenant_id=tenant_id, token=credentials.get("config").get("api_key"))204 response = requests.post(205 "https://adaptivecrawlstatus-kir3wx7b3a-uc.a.run.app",206 headers={"Content-Type": "application/json", "Authorization": f"Bearer {api_key}"},207 json={"taskId": job_id},208 )209 data = response.json().get("data", {})210 if data.get("status") != "completed":211 raise ValueError("Crawl job is not completed")212 213 response = requests.post(214 "https://adaptivecrawlstatus-kir3wx7b3a-uc.a.run.app",215 headers={"Content-Type": "application/json", "Authorization": f"Bearer {api_key}"},216 json={"taskId": job_id, "urls": list(data.get("processed", {}).keys())},217 )218 data = response.json().get("data", {})219 for item in data.get("processed", {}).values():220 if item.get("data", {}).get("url") == url:221 return item.get("data", {})222 else:223 raise ValueError("Invalid provider")224 225 @classmethod226 def get_scrape_url_data(cls, provider: str, url: str, tenant_id: str, only_main_content: bool) -> dict | None:227 credentials = ApiKeyAuthService.get_auth_credentials(tenant_id, "website", provider)228 if provider == "firecrawl":229 # decrypt api_key230 api_key = encrypter.decrypt_token(tenant_id=tenant_id, token=credentials.get("config").get("api_key"))231 firecrawl_app = FirecrawlApp(api_key=api_key, base_url=credentials.get("config").get("base_url", None))232 params = {"pageOptions": {"onlyMainContent": only_main_content, "includeHtml": False}}233 result = firecrawl_app.scrape_url(url, params)234 return result235 else:236 raise ValueError("Invalid provider")237 