Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
robotparser.py289 linesDownload Raw Back to urllib
1""" robotparser.py2 3    Copyright (C) 2000  Bastian Kleineidam4 5    You can choose between two licenses when using this package:6    1) GNU GPLv27    2) PSF license for Python 2.28 9    The robots.txt Exclusion Protocol is implemented as specified in10    http://www.robotstxt.org/norobots-rfc.txt11"""12 13import collections14import re15import urllib.error16import urllib.parse17import urllib.request18 19__all__ = ["RobotFileParser"]20 21RequestRate = collections.namedtuple("RequestRate", "requests seconds")22 23 24def normalize(path):25    unquoted = urllib.parse.unquote(path, errors='surrogateescape')26    return urllib.parse.quote(unquoted, errors='surrogateescape')27 28def normalize_path(path):29    path, sep, query = path.partition('?')30    path = normalize(path)31    if sep:32        query = re.sub(r'[^=&]+', lambda m: normalize(m[0]), query)33        path += '?' + query34    return path35 36 37class RobotFileParser:38    """ This class provides a set of methods to read, parse and answer39    questions about a single robots.txt file.40 41    """42 43    def __init__(self, url=''):44        self.entries = []45        self.sitemaps = []46        self.default_entry = None47        self.disallow_all = False48        self.allow_all = False49        self.set_url(url)50        self.last_checked = 051 52    def mtime(self):53        """Returns the time the robots.txt file was last fetched.54 55        This is useful for long-running web spiders that need to56        check for new robots.txt files periodically.57 58        """59        return self.last_checked60 61    def modified(self):62        """Sets the time the robots.txt file was last fetched to the63        current time.64 65        """66        import time67        self.last_checked = time.time()68 69    def set_url(self, url):70        """Sets the URL referring to a robots.txt file."""71        self.url = url72        self.host, self.path = urllib.parse.urlsplit(url)[1:3]73 74    def read(self):75        """Reads the robots.txt URL and feeds it to the parser."""76        try:77            f = urllib.request.urlopen(self.url)78        except urllib.error.HTTPError as err:79            if err.code in (401, 403):80                self.disallow_all = True81            elif err.code >= 400 and err.code < 500:82                self.allow_all = True83            err.close()84        else:85            raw = f.read()86            self.parse(raw.decode("utf-8", "surrogateescape").splitlines())87 88    def _add_entry(self, entry):89        if "*" in entry.useragents:90            # the default entry is considered last91            if self.default_entry is None:92                # the first default entry wins93                self.default_entry = entry94        else:95            self.entries.append(entry)96 97    def parse(self, lines):98        """Parse the input lines from a robots.txt file.99 100        We allow that a user-agent: line is not preceded by101        one or more blank lines.102        """103        # states:104        #   0: start state105        #   1: saw user-agent line106        #   2: saw an allow or disallow line107        state = 0108        entry = Entry()109 110        self.modified()111        for line in lines:112            if not line:113                if state == 1:114                    entry = Entry()115                    state = 0116                elif state == 2:117                    self._add_entry(entry)118                    entry = Entry()119                    state = 0120            # remove optional comment and strip line121            i = line.find('#')122            if i >= 0:123                line = line[:i]124            line = line.strip()125            if not line:126                continue127            line = line.split(':', 1)128            if len(line) == 2:129                line[0] = line[0].strip().lower()130                line[1] = line[1].strip()131                if line[0] == "user-agent":132                    if state == 2:133                        self._add_entry(entry)134                        entry = Entry()135                    entry.useragents.append(line[1])136                    state = 1137                elif line[0] == "disallow":138                    if state != 0:139                        entry.rulelines.append(RuleLine(line[1], False))140                        state = 2141                elif line[0] == "allow":142                    if state != 0:143                        entry.rulelines.append(RuleLine(line[1], True))144                        state = 2145                elif line[0] == "crawl-delay":146                    if state != 0:147                        # before trying to convert to int we need to make148                        # sure that robots.txt has valid syntax otherwise149                        # it will crash150                        if line[1].strip().isdigit():151                            entry.delay = int(line[1])152                        state = 2153                elif line[0] == "request-rate":154                    if state != 0:155                        numbers = line[1].split('/')156                        # check if all values are sane157                        if (len(numbers) == 2 and numbers[0].strip().isdigit()158                            and numbers[1].strip().isdigit()):159                            entry.req_rate = RequestRate(int(numbers[0]), int(numbers[1]))160                        state = 2161                elif line[0] == "sitemap":162                    # According to http://www.sitemaps.org/protocol.html163                    # "This directive is independent of the user-agent line,164                    #  so it doesn't matter where you place it in your file."165                    # Therefore we do not change the state of the parser.166                    self.sitemaps.append(line[1])167        if state == 2:168            self._add_entry(entry)169 170    def can_fetch(self, useragent, url):171        """using the parsed robots.txt decide if useragent can fetch url"""172        if self.disallow_all:173            return False174        if self.allow_all:175            return True176        # Until the robots.txt file has been read or found not177        # to exist, we must assume that no url is allowable.178        # This prevents false positives when a user erroneously179        # calls can_fetch() before calling read().180        if not self.last_checked:181            return False182        # search for given user agent matches183        # the first match counts184        # TODO: The private API is used in order to preserve an empty query.185        # This is temporary until the public API starts supporting this feature.186        parsed_url = urllib.parse._urlsplit(url, '')187        url = urllib.parse._urlunsplit(None, None, *parsed_url[2:])188        url = normalize_path(url)189        if not url:190            url = "/"191        for entry in self.entries:192            if entry.applies_to(useragent):193                return entry.allowance(url)194        # try the default entry last195        if self.default_entry:196            return self.default_entry.allowance(url)197        # agent not found ==> access granted198        return True199 200    def crawl_delay(self, useragent):201        if not self.mtime():202            return None203        for entry in self.entries:204            if entry.applies_to(useragent):205                return entry.delay206        if self.default_entry:207            return self.default_entry.delay208        return None209 210    def request_rate(self, useragent):211        if not self.mtime():212            return None213        for entry in self.entries:214            if entry.applies_to(useragent):215                return entry.req_rate216        if self.default_entry:217            return self.default_entry.req_rate218        return None219 220    def site_maps(self):221        if not self.sitemaps:222            return None223        return self.sitemaps224 225    def __str__(self):226        entries = self.entries227        if self.default_entry is not None:228            entries = entries + [self.default_entry]229        return '\n\n'.join(map(str, entries))230 231class RuleLine:232    """A rule line is a single "Allow:" (allowance==True) or "Disallow:"233       (allowance==False) followed by a path."""234    def __init__(self, path, allowance):235        if path == '' and not allowance:236            # an empty value means allow all237            allowance = True238        self.path = normalize_path(path)239        self.allowance = allowance240 241    def applies_to(self, filename):242        return self.path == "*" or filename.startswith(self.path)243 244    def __str__(self):245        return ("Allow" if self.allowance else "Disallow") + ": " + self.path246 247 248class Entry:249    """An entry has one or more user-agents and zero or more rulelines"""250    def __init__(self):251        self.useragents = []252        self.rulelines = []253        self.delay = None254        self.req_rate = None255 256    def __str__(self):257        ret = []258        for agent in self.useragents:259            ret.append(f"User-agent: {agent}")260        if self.delay is not None:261            ret.append(f"Crawl-delay: {self.delay}")262        if self.req_rate is not None:263            rate = self.req_rate264            ret.append(f"Request-rate: {rate.requests}/{rate.seconds}")265        ret.extend(map(str, self.rulelines))266        return '\n'.join(ret)267 268    def applies_to(self, useragent):269        """check if this entry applies to the specified agent"""270        # split the name token and make it lower case271        useragent = useragent.split("/")[0].lower()272        for agent in self.useragents:273            if agent == '*':274                # we have the catch-all agent275                return True276            agent = agent.lower()277            if agent in useragent:278                return True279        return False280 281    def allowance(self, filename):282        """Preconditions:283        - our agent applies to this entry284        - filename is URL encoded"""285        for line in self.rulelines:286            if line.applies_to(filename):287                return line.allowance288        return True289 
codekingpro/portable-devtools · Team Ai