codekingpro/portable-devtools
114k
1""" robotparser.py2 3 Copyright (C) 2000 Bastian Kleineidam4 5 You can choose between two licenses when using this package:6 1) GNU GPLv27 2) PSF license for Python 2.28 9 The robots.txt Exclusion Protocol is implemented as specified in10 http://www.robotstxt.org/norobots-rfc.txt11"""12 13import collections14import re15import urllib.error16import urllib.parse17import urllib.request18 19__all__ = ["RobotFileParser"]20 21RequestRate = collections.namedtuple("RequestRate", "requests seconds")22 23 24def normalize(path):25 unquoted = urllib.parse.unquote(path, errors='surrogateescape')26 return urllib.parse.quote(unquoted, errors='surrogateescape')27 28def normalize_path(path):29 path, sep, query = path.partition('?')30 path = normalize(path)31 if sep:32 query = re.sub(r'[^=&]+', lambda m: normalize(m[0]), query)33 path += '?' + query34 return path35 36 37class RobotFileParser:38 """ This class provides a set of methods to read, parse and answer39 questions about a single robots.txt file.40 41 """42 43 def __init__(self, url=''):44 self.entries = []45 self.sitemaps = []46 self.default_entry = None47 self.disallow_all = False48 self.allow_all = False49 self.set_url(url)50 self.last_checked = 051 52 def mtime(self):53 """Returns the time the robots.txt file was last fetched.54 55 This is useful for long-running web spiders that need to56 check for new robots.txt files periodically.57 58 """59 return self.last_checked60 61 def modified(self):62 """Sets the time the robots.txt file was last fetched to the63 current time.64 65 """66 import time67 self.last_checked = time.time()68 69 def set_url(self, url):70 """Sets the URL referring to a robots.txt file."""71 self.url = url72 self.host, self.path = urllib.parse.urlsplit(url)[1:3]73 74 def read(self):75 """Reads the robots.txt URL and feeds it to the parser."""76 try:77 f = urllib.request.urlopen(self.url)78 except urllib.error.HTTPError as err:79 if err.code in (401, 403):80 self.disallow_all = True81 elif err.code >= 400 and err.code < 500:82 self.allow_all = True83 err.close()84 else:85 raw = f.read()86 self.parse(raw.decode("utf-8", "surrogateescape").splitlines())87 88 def _add_entry(self, entry):89 if "*" in entry.useragents:90 # the default entry is considered last91 if self.default_entry is None:92 # the first default entry wins93 self.default_entry = entry94 else:95 self.entries.append(entry)96 97 def parse(self, lines):98 """Parse the input lines from a robots.txt file.99 100 We allow that a user-agent: line is not preceded by101 one or more blank lines.102 """103 # states:104 # 0: start state105 # 1: saw user-agent line106 # 2: saw an allow or disallow line107 state = 0108 entry = Entry()109 110 self.modified()111 for line in lines:112 if not line:113 if state == 1:114 entry = Entry()115 state = 0116 elif state == 2:117 self._add_entry(entry)118 entry = Entry()119 state = 0120 # remove optional comment and strip line121 i = line.find('#')122 if i >= 0:123 line = line[:i]124 line = line.strip()125 if not line:126 continue127 line = line.split(':', 1)128 if len(line) == 2:129 line[0] = line[0].strip().lower()130 line[1] = line[1].strip()131 if line[0] == "user-agent":132 if state == 2:133 self._add_entry(entry)134 entry = Entry()135 entry.useragents.append(line[1])136 state = 1137 elif line[0] == "disallow":138 if state != 0:139 entry.rulelines.append(RuleLine(line[1], False))140 state = 2141 elif line[0] == "allow":142 if state != 0:143 entry.rulelines.append(RuleLine(line[1], True))144 state = 2145 elif line[0] == "crawl-delay":146 if state != 0:147 # before trying to convert to int we need to make148 # sure that robots.txt has valid syntax otherwise149 # it will crash150 if line[1].strip().isdigit():151 entry.delay = int(line[1])152 state = 2153 elif line[0] == "request-rate":154 if state != 0:155 numbers = line[1].split('/')156 # check if all values are sane157 if (len(numbers) == 2 and numbers[0].strip().isdigit()158 and numbers[1].strip().isdigit()):159 entry.req_rate = RequestRate(int(numbers[0]), int(numbers[1]))160 state = 2161 elif line[0] == "sitemap":162 # According to http://www.sitemaps.org/protocol.html163 # "This directive is independent of the user-agent line,164 # so it doesn't matter where you place it in your file."165 # Therefore we do not change the state of the parser.166 self.sitemaps.append(line[1])167 if state == 2:168 self._add_entry(entry)169 170 def can_fetch(self, useragent, url):171 """using the parsed robots.txt decide if useragent can fetch url"""172 if self.disallow_all:173 return False174 if self.allow_all:175 return True176 # Until the robots.txt file has been read or found not177 # to exist, we must assume that no url is allowable.178 # This prevents false positives when a user erroneously179 # calls can_fetch() before calling read().180 if not self.last_checked:181 return False182 # search for given user agent matches183 # the first match counts184 # TODO: The private API is used in order to preserve an empty query.185 # This is temporary until the public API starts supporting this feature.186 parsed_url = urllib.parse._urlsplit(url, '')187 url = urllib.parse._urlunsplit(None, None, *parsed_url[2:])188 url = normalize_path(url)189 if not url:190 url = "/"191 for entry in self.entries:192 if entry.applies_to(useragent):193 return entry.allowance(url)194 # try the default entry last195 if self.default_entry:196 return self.default_entry.allowance(url)197 # agent not found ==> access granted198 return True199 200 def crawl_delay(self, useragent):201 if not self.mtime():202 return None203 for entry in self.entries:204 if entry.applies_to(useragent):205 return entry.delay206 if self.default_entry:207 return self.default_entry.delay208 return None209 210 def request_rate(self, useragent):211 if not self.mtime():212 return None213 for entry in self.entries:214 if entry.applies_to(useragent):215 return entry.req_rate216 if self.default_entry:217 return self.default_entry.req_rate218 return None219 220 def site_maps(self):221 if not self.sitemaps:222 return None223 return self.sitemaps224 225 def __str__(self):226 entries = self.entries227 if self.default_entry is not None:228 entries = entries + [self.default_entry]229 return '\n\n'.join(map(str, entries))230 231class RuleLine:232 """A rule line is a single "Allow:" (allowance==True) or "Disallow:"233 (allowance==False) followed by a path."""234 def __init__(self, path, allowance):235 if path == '' and not allowance:236 # an empty value means allow all237 allowance = True238 self.path = normalize_path(path)239 self.allowance = allowance240 241 def applies_to(self, filename):242 return self.path == "*" or filename.startswith(self.path)243 244 def __str__(self):245 return ("Allow" if self.allowance else "Disallow") + ": " + self.path246 247 248class Entry:249 """An entry has one or more user-agents and zero or more rulelines"""250 def __init__(self):251 self.useragents = []252 self.rulelines = []253 self.delay = None254 self.req_rate = None255 256 def __str__(self):257 ret = []258 for agent in self.useragents:259 ret.append(f"User-agent: {agent}")260 if self.delay is not None:261 ret.append(f"Crawl-delay: {self.delay}")262 if self.req_rate is not None:263 rate = self.req_rate264 ret.append(f"Request-rate: {rate.requests}/{rate.seconds}")265 ret.extend(map(str, self.rulelines))266 return '\n'.join(ret)267 268 def applies_to(self, useragent):269 """check if this entry applies to the specified agent"""270 # split the name token and make it lower case271 useragent = useragent.split("/")[0].lower()272 for agent in self.useragents:273 if agent == '*':274 # we have the catch-all agent275 return True276 agent = agent.lower()277 if agent in useragent:278 return True279 return False280 281 def allowance(self, filename):282 """Preconditions:283 - our agent applies to this entry284 - filename is URL encoded"""285 for line in self.rulelines:286 if line.applies_to(filename):287 return line.allowance288 return True289 