codekingpro/portable-devtools
114k
1"""The Tab Nanny despises ambiguous indentation. She knows no mercy.2 3tabnanny -- Detection of ambiguous indentation4 5For the time being this module is intended to be called as a script.6However it is possible to import it into an IDE and use the function7check() described below.8 9Warning: The API provided by this module is likely to change in future10releases; such changes may not be backward compatible.11"""12 13# Released to the public domain, by Tim Peters, 15 April 1998.14 15# XXX Note: this is now a standard library module.16# XXX The API needs to undergo changes however; the current code is too17# XXX script-like. This will be addressed later.18 19__version__ = "6"20 21import os22import sys23import tokenize24 25__all__ = ["check", "NannyNag", "process_tokens"]26 27verbose = 028filename_only = 029 30def errprint(*args):31 sep = ""32 for arg in args:33 sys.stderr.write(sep + str(arg))34 sep = " "35 sys.stderr.write("\n")36 sys.exit(1)37 38def main():39 import getopt40 41 global verbose, filename_only42 try:43 opts, args = getopt.getopt(sys.argv[1:], "qv")44 except getopt.error as msg:45 errprint(msg)46 for o, a in opts:47 if o == '-q':48 filename_only = filename_only + 149 if o == '-v':50 verbose = verbose + 151 if not args:52 errprint("Usage:", sys.argv[0], "[-v] file_or_directory ...")53 for arg in args:54 check(arg)55 56class NannyNag(Exception):57 """58 Raised by process_tokens() if detecting an ambiguous indent.59 Captured and handled in check().60 """61 def __init__(self, lineno, msg, line):62 self.lineno, self.msg, self.line = lineno, msg, line63 def get_lineno(self):64 return self.lineno65 def get_msg(self):66 return self.msg67 def get_line(self):68 return self.line69 70def check(file):71 """check(file_or_dir)72 73 If file_or_dir is a directory and not a symbolic link, then recursively74 descend the directory tree named by file_or_dir, checking all .py files75 along the way. If file_or_dir is an ordinary Python source file, it is76 checked for whitespace related problems. The diagnostic messages are77 written to standard output using the print statement.78 """79 80 if os.path.isdir(file) and not os.path.islink(file):81 if verbose:82 print("%r: listing directory" % (file,))83 names = os.listdir(file)84 for name in names:85 fullname = os.path.join(file, name)86 if (os.path.isdir(fullname) and87 not os.path.islink(fullname) or88 os.path.normcase(name[-3:]) == ".py"):89 check(fullname)90 return91 92 try:93 f = tokenize.open(file)94 except OSError as msg:95 errprint("%r: I/O Error: %s" % (file, msg))96 return97 98 if verbose > 1:99 print("checking %r ..." % file)100 101 try:102 process_tokens(tokenize.generate_tokens(f.readline))103 104 except tokenize.TokenError as msg:105 errprint("%r: Token Error: %s" % (file, msg))106 return107 108 except IndentationError as msg:109 errprint("%r: Indentation Error: %s" % (file, msg))110 return111 112 except SyntaxError as msg:113 errprint("%r: Syntax Error: %s" % (file, msg))114 return115 116 except NannyNag as nag:117 badline = nag.get_lineno()118 line = nag.get_line()119 if verbose:120 print("%r: *** Line %d: trouble in tab city! ***" % (file, badline))121 print("offending line: %r" % (line,))122 print(nag.get_msg())123 else:124 if ' ' in file: file = '"' + file + '"'125 if filename_only: print(file)126 else: print(file, badline, repr(line))127 return128 129 finally:130 f.close()131 132 if verbose:133 print("%r: Clean bill of health." % (file,))134 135class Whitespace:136 # the characters used for space and tab137 S, T = ' \t'138 139 # members:140 # raw141 # the original string142 # n143 # the number of leading whitespace characters in raw144 # nt145 # the number of tabs in raw[:n]146 # norm147 # the normal form as a pair (count, trailing), where:148 # count149 # a tuple such that raw[:n] contains count[i]150 # instances of S * i + T151 # trailing152 # the number of trailing spaces in raw[:n]153 # It's A Theorem that m.indent_level(t) ==154 # n.indent_level(t) for all t >= 1 iff m.norm == n.norm.155 # is_simple156 # true iff raw[:n] is of the form (T*)(S*)157 158 def __init__(self, ws):159 self.raw = ws160 S, T = Whitespace.S, Whitespace.T161 count = []162 b = n = nt = 0163 for ch in self.raw:164 if ch == S:165 n = n + 1166 b = b + 1167 elif ch == T:168 n = n + 1169 nt = nt + 1170 if b >= len(count):171 count = count + [0] * (b - len(count) + 1)172 count[b] = count[b] + 1173 b = 0174 else:175 break176 self.n = n177 self.nt = nt178 self.norm = tuple(count), b179 self.is_simple = len(count) <= 1180 181 # return length of longest contiguous run of spaces (whether or not182 # preceding a tab)183 def longest_run_of_spaces(self):184 count, trailing = self.norm185 return max(len(count)-1, trailing)186 187 def indent_level(self, tabsize):188 # count, il = self.norm189 # for i in range(len(count)):190 # if count[i]:191 # il = il + (i//tabsize + 1)*tabsize * count[i]192 # return il193 194 # quicker:195 # il = trailing + sum (i//ts + 1)*ts*count[i] =196 # trailing + ts * sum (i//ts + 1)*count[i] =197 # trailing + ts * sum i//ts*count[i] + count[i] =198 # trailing + ts * [(sum i//ts*count[i]) + (sum count[i])] =199 # trailing + ts * [(sum i//ts*count[i]) + num_tabs]200 # and note that i//ts*count[i] is 0 when i < ts201 202 count, trailing = self.norm203 il = 0204 for i in range(tabsize, len(count)):205 il = il + i//tabsize * count[i]206 return trailing + tabsize * (il + self.nt)207 208 # return true iff self.indent_level(t) == other.indent_level(t)209 # for all t >= 1210 def equal(self, other):211 return self.norm == other.norm212 213 # return a list of tuples (ts, i1, i2) such that214 # i1 == self.indent_level(ts) != other.indent_level(ts) == i2.215 # Intended to be used after not self.equal(other) is known, in which216 # case it will return at least one witnessing tab size.217 def not_equal_witness(self, other):218 n = max(self.longest_run_of_spaces(),219 other.longest_run_of_spaces()) + 1220 a = []221 for ts in range(1, n+1):222 if self.indent_level(ts) != other.indent_level(ts):223 a.append( (ts,224 self.indent_level(ts),225 other.indent_level(ts)) )226 return a227 228 # Return True iff self.indent_level(t) < other.indent_level(t)229 # for all t >= 1.230 # The algorithm is due to Vincent Broman.231 # Easy to prove it's correct.232 # XXXpost that.233 # Trivial to prove n is sharp (consider T vs ST).234 # Unknown whether there's a faster general way. I suspected so at235 # first, but no longer.236 # For the special (but common!) case where M and N are both of the237 # form (T*)(S*), M.less(N) iff M.len() < N.len() and238 # M.num_tabs() <= N.num_tabs(). Proof is easy but kinda long-winded.239 # XXXwrite that up.240 # Note that M is of the form (T*)(S*) iff len(M.norm[0]) <= 1.241 def less(self, other):242 if self.n >= other.n:243 return False244 if self.is_simple and other.is_simple:245 return self.nt <= other.nt246 n = max(self.longest_run_of_spaces(),247 other.longest_run_of_spaces()) + 1248 # the self.n >= other.n test already did it for ts=1249 for ts in range(2, n+1):250 if self.indent_level(ts) >= other.indent_level(ts):251 return False252 return True253 254 # return a list of tuples (ts, i1, i2) such that255 # i1 == self.indent_level(ts) >= other.indent_level(ts) == i2.256 # Intended to be used after not self.less(other) is known, in which257 # case it will return at least one witnessing tab size.258 def not_less_witness(self, other):259 n = max(self.longest_run_of_spaces(),260 other.longest_run_of_spaces()) + 1261 a = []262 for ts in range(1, n+1):263 if self.indent_level(ts) >= other.indent_level(ts):264 a.append( (ts,265 self.indent_level(ts),266 other.indent_level(ts)) )267 return a268 269def format_witnesses(w):270 firsts = (str(tup[0]) for tup in w)271 prefix = "at tab size"272 if len(w) > 1:273 prefix = prefix + "s"274 return prefix + " " + ', '.join(firsts)275 276def process_tokens(tokens):277 try:278 _process_tokens(tokens)279 except TabError as e:280 raise NannyNag(e.lineno, e.msg, e.text)281 282def _process_tokens(tokens):283 INDENT = tokenize.INDENT284 DEDENT = tokenize.DEDENT285 NEWLINE = tokenize.NEWLINE286 JUNK = tokenize.COMMENT, tokenize.NL287 indents = [Whitespace("")]288 check_equal = 0289 290 for (type, token, start, end, line) in tokens:291 if type == NEWLINE:292 # a program statement, or ENDMARKER, will eventually follow,293 # after some (possibly empty) run of tokens of the form294 # (NL | COMMENT)* (INDENT | DEDENT+)?295 # If an INDENT appears, setting check_equal is wrong, and will296 # be undone when we see the INDENT.297 check_equal = 1298 299 elif type == INDENT:300 check_equal = 0301 thisguy = Whitespace(token)302 if not indents[-1].less(thisguy):303 witness = indents[-1].not_less_witness(thisguy)304 msg = "indent not greater e.g. " + format_witnesses(witness)305 raise NannyNag(start[0], msg, line)306 indents.append(thisguy)307 308 elif type == DEDENT:309 # there's nothing we need to check here! what's important is310 # that when the run of DEDENTs ends, the indentation of the311 # program statement (or ENDMARKER) that triggered the run is312 # equal to what's left at the top of the indents stack313 314 # Ouch! This assert triggers if the last line of the source315 # is indented *and* lacks a newline -- then DEDENTs pop out316 # of thin air.317 # assert check_equal # else no earlier NEWLINE, or an earlier INDENT318 check_equal = 1319 320 del indents[-1]321 322 elif check_equal and type not in JUNK:323 # this is the first "real token" following a NEWLINE, so it324 # must be the first token of the next program statement, or an325 # ENDMARKER; the "line" argument exposes the leading whitespace326 # for this statement; in the case of ENDMARKER, line is an empty327 # string, so will properly match the empty string with which the328 # "indents" stack was seeded329 check_equal = 0330 thisguy = Whitespace(line)331 if not indents[-1].equal(thisguy):332 witness = indents[-1].not_equal_witness(thisguy)333 msg = "indent not equal e.g. " + format_witnesses(witness)334 raise NannyNag(start[0], msg, line)335 336 337if __name__ == '__main__':338 main()339 