codekingpro/portable-devtools
114k
1from typing import List, Optional2 3import requests4from langchain_core.documents import Document5 6from langchain_community.document_loaders.base import BaseLoader7from langchain_community.document_loaders.web_base import WebBaseLoader8 9IFIXIT_BASE_URL = "https://www.ifixit.com/api/2.0"10 11 12class IFixitLoader(BaseLoader):13 """Load `iFixit` repair guides, device wikis and answers.14 15 iFixit is the largest, open repair community on the web. The site contains nearly16 100k repair manuals, 200k Questions & Answers on 42k devices, and all the data is17 licensed under CC-BY.18 19 This loader will allow you to download the text of a repair guide, text of Q&A's20 and wikis from devices on iFixit using their open APIs and web scraping.21 """22 23 def __init__(self, web_path: str):24 """Initialize with a web path."""25 if not web_path.startswith("https://www.ifixit.com"):26 raise ValueError("web path must start with 'https://www.ifixit.com'")27 28 path = web_path.replace("https://www.ifixit.com", "")29 30 allowed_paths = ["/Device", "/Guide", "/Answers", "/Teardown"]31 32 """ TODO: Add /Wiki """33 if not any(path.startswith(allowed_path) for allowed_path in allowed_paths):34 raise ValueError(35 "web path must start with /Device, /Guide, /Teardown or /Answers"36 )37 38 pieces = [x for x in path.split("/") if x]39 40 """Teardowns are just guides by a different name"""41 self.page_type = pieces[0] if pieces[0] != "Teardown" else "Guide"42 43 if self.page_type == "Guide" or self.page_type == "Answers":44 self.id = pieces[2]45 else:46 self.id = pieces[1]47 48 self.web_path = web_path49 50 def load(self) -> List[Document]:51 if self.page_type == "Device":52 return self.load_device()53 elif self.page_type == "Guide" or self.page_type == "Teardown":54 return self.load_guide()55 elif self.page_type == "Answers":56 return self.load_questions_and_answers()57 else:58 raise ValueError("Unknown page type: " + self.page_type)59 60 @staticmethod61 def load_suggestions(query: str = "", doc_type: str = "all") -> List[Document]:62 """Load suggestions.63 64 Args:65 query: A query string66 doc_type: The type of document to search for. Can be one of "all",67 "device", "guide", "teardown", "answer", "wiki".68 69 Returns:70 71 """72 res = requests.get(73 IFIXIT_BASE_URL + "/suggest/" + query + "?doctypes=" + doc_type74 )75 76 if res.status_code != 200:77 raise ValueError(78 'Could not load suggestions for "' + query + '"\n' + res.json()79 )80 81 data = res.json()82 83 results = data["results"]84 output = []85 86 for result in results:87 try:88 loader = IFixitLoader(result["url"])89 if loader.page_type == "Device":90 output += loader.load_device(include_guides=False)91 else:92 output += loader.load()93 except ValueError:94 continue95 96 return output97 98 def load_questions_and_answers(99 self, url_override: Optional[str] = None100 ) -> List[Document]:101 """Load a list of questions and answers.102 103 Args:104 url_override: A URL to override the default URL.105 106 Returns: List[Document]107 108 """109 loader = WebBaseLoader(self.web_path if url_override is None else url_override)110 soup = loader.scrape()111 112 output = []113 114 title = soup.find("h1", "post-title").text115 116 output.append("# " + title)117 output.append(soup.select_one(".post-content .post-text").text.strip())118 119 answersHeader = soup.find("div", "post-answers-header")120 if answersHeader:121 output.append("\n## " + answersHeader.text.strip())122 123 for answer in soup.select(".js-answers-list .post.post-answer"):124 if answer.has_attr("itemprop") and "acceptedAnswer" in answer["itemprop"]:125 output.append("\n### Accepted Answer")126 elif "post-helpful" in answer["class"]:127 output.append("\n### Most Helpful Answer")128 else:129 output.append("\n### Other Answer")130 131 output += [132 a.text.strip() for a in answer.select(".post-content .post-text")133 ]134 output.append("\n")135 136 text = "\n".join(output).strip()137 138 metadata = {"source": self.web_path, "title": title}139 140 return [Document(page_content=text, metadata=metadata)]141 142 def load_device(143 self, url_override: Optional[str] = None, include_guides: bool = True144 ) -> List[Document]:145 """Loads a device146 147 Args:148 url_override: A URL to override the default URL.149 include_guides: Whether to include guides linked to from the device.150 Defaults to True.151 152 Returns:153 154 """155 documents = []156 if url_override is None:157 url = IFIXIT_BASE_URL + "/wikis/CATEGORY/" + self.id158 else:159 url = url_override160 161 res = requests.get(url)162 data = res.json()163 text = "\n".join(164 [165 data[key]166 for key in ["title", "description", "contents_raw"]167 if key in data168 ]169 ).strip()170 171 metadata = {"source": self.web_path, "title": data["title"]}172 documents.append(Document(page_content=text, metadata=metadata))173 174 if include_guides:175 """Load and return documents for each guide linked to from the device"""176 guide_urls = [guide["url"] for guide in data["guides"]]177 for guide_url in guide_urls:178 documents.append(IFixitLoader(guide_url).load()[0])179 180 return documents181 182 def load_guide(self, url_override: Optional[str] = None) -> List[Document]:183 """Load a guide184 185 Args:186 url_override: A URL to override the default URL.187 188 Returns: List[Document]189 190 """191 if url_override is None:192 url = IFIXIT_BASE_URL + "/guides/" + self.id193 else:194 url = url_override195 196 res = requests.get(url)197 198 if res.status_code != 200:199 raise ValueError(200 "Could not load guide: " + self.web_path + "\n" + res.json()201 )202 203 data = res.json()204 205 doc_parts = ["# " + data["title"], data["introduction_raw"]]206 207 doc_parts.append("\n\n###Tools Required:")208 if len(data["tools"]) == 0:209 doc_parts.append("\n - None")210 else:211 for tool in data["tools"]:212 doc_parts.append("\n - " + tool["text"])213 214 doc_parts.append("\n\n###Parts Required:")215 if len(data["parts"]) == 0:216 doc_parts.append("\n - None")217 else:218 for part in data["parts"]:219 doc_parts.append("\n - " + part["text"])220 221 for row in data["steps"]:222 doc_parts.append(223 "\n\n## "224 + (225 row["title"]226 if row["title"] != ""227 else "Step {}".format(row["orderby"])228 )229 )230 231 for line in row["lines"]:232 doc_parts.append(line["text_raw"])233 234 doc_parts.append(data["conclusion_raw"])235 236 text = "\n".join(doc_parts)237 238 metadata = {"source": self.web_path, "title": data["title"]}239 240 return [Document(page_content=text, metadata=metadata)]241 