codekingpro/portable-devtools
114k
1from __future__ import annotations
2
3import argparse
4import sys
5import typing
6from json import dumps
7from os.path import abspath, basename, dirname, join, realpath
8from platform import python_version
9from unicodedata import unidata_version
10
11import charset_normalizer.md as md_module
12from charset_normalizer import from_fp
13from charset_normalizer.models import CliDetectionResult
14from charset_normalizer.version import __version__
15
16
17def query_yes_no(question: str, default: str = "yes") -> bool: # Defensive:
18 """Ask a yes/no question via input() and return the answer as a bool."""
19 prompt = " [Y/n] " if default == "yes" else " [y/N] "
20
21 while True:
22 choice = input(question + prompt).strip().lower()
23 if not choice:
24 return default == "yes"
25 if choice in ("y", "yes"):
26 return True
27 if choice in ("n", "no"):
28 return False
29 print("Please respond with 'y' or 'n'.")
30
31
32class FileType:
33 """Factory for creating file object types
34
35 Instances of FileType are typically passed as type= arguments to the
36 ArgumentParser add_argument() method.
37
38 Keyword Arguments:
39 - mode -- A string indicating how the file is to be opened. Accepts the
40 same values as the builtin open() function.
41 - bufsize -- The file's desired buffer size. Accepts the same values as
42 the builtin open() function.
43 - encoding -- The file's encoding. Accepts the same values as the
44 builtin open() function.
45 - errors -- A string indicating how encoding and decoding errors are to
46 be handled. Accepts the same value as the builtin open() function.
47
48 Backported from CPython 3.12
49 """
50
51 def __init__(
52 self,
53 mode: str = "r",
54 bufsize: int = -1,
55 encoding: str | None = None,
56 errors: str | None = None,
57 ):
58 self._mode = mode
59 self._bufsize = bufsize
60 self._encoding = encoding
61 self._errors = errors
62
63 def __call__(self, string: str) -> typing.IO: # type: ignore[type-arg]
64 # the special argument "-" means sys.std{in,out}
65 if string == "-":
66 if "r" in self._mode:
67 return sys.stdin.buffer if "b" in self._mode else sys.stdin
68 elif any(c in self._mode for c in "wax"):
69 return sys.stdout.buffer if "b" in self._mode else sys.stdout
70 else:
71 msg = f'argument "-" with mode {self._mode}'
72 raise ValueError(msg)
73
74 # all other arguments are used as file names
75 try:
76 return open(string, self._mode, self._bufsize, self._encoding, self._errors)
77 except OSError as e:
78 message = f"can't open '{string}': {e}"
79 raise argparse.ArgumentTypeError(message)
80
81 def __repr__(self) -> str:
82 args = self._mode, self._bufsize
83 kwargs = [("encoding", self._encoding), ("errors", self._errors)]
84 args_str = ", ".join(
85 [repr(arg) for arg in args if arg != -1]
86 + [f"{kw}={arg!r}" for kw, arg in kwargs if arg is not None]
87 )
88 return f"{type(self).__name__}({args_str})"
89
90
91def cli_detect(argv: list[str] | None = None) -> int:
92 """
93 CLI assistant using ARGV and ArgumentParser
94 :param argv:
95 :return: 0 if everything is fine, anything else equal trouble
96 """
97 parser = argparse.ArgumentParser(
98 description="The Real First Universal Charset Detector. "
99 "Discover originating encoding used on text file. "
100 "Normalize text to unicode."
101 )
102
103 parser.add_argument(
104 "files", type=FileType("rb"), nargs="+", help="File(s) to be analysed"
105 )
106 parser.add_argument(
107 "-v",
108 "--verbose",
109 action="store_true",
110 default=False,
111 dest="verbose",
112 help="Display complementary information about file if any. "
113 "Stdout will contain logs about the detection process.",
114 )
115 parser.add_argument(
116 "-a",
117 "--with-alternative",
118 action="store_true",
119 default=False,
120 dest="alternatives",
121 help="Output complementary possibilities if any. Top-level JSON WILL be a list.",
122 )
123 parser.add_argument(
124 "-n",
125 "--normalize",
126 action="store_true",
127 default=False,
128 dest="normalize",
129 help="Permit to normalize input file. If not set, program does not write anything.",
130 )
131 parser.add_argument(
132 "-m",
133 "--minimal",
134 action="store_true",
135 default=False,
136 dest="minimal",
137 help="Only output the charset detected to STDOUT. Disabling JSON output.",
138 )
139 parser.add_argument(
140 "-r",
141 "--replace",
142 action="store_true",
143 default=False,
144 dest="replace",
145 help="Replace file when trying to normalize it instead of creating a new one.",
146 )
147 parser.add_argument(
148 "-f",
149 "--force",
150 action="store_true",
151 default=False,
152 dest="force",
153 help="Replace file without asking if you are sure, use this flag with caution.",
154 )
155 parser.add_argument(
156 "-i",
157 "--no-preemptive",
158 action="store_true",
159 default=False,
160 dest="no_preemptive",
161 help="Disable looking at a charset declaration to hint the detector.",
162 )
163 parser.add_argument(
164 "-t",
165 "--threshold",
166 action="store",
167 default=0.2,
168 type=float,
169 dest="threshold",
170 help="Define a custom maximum amount of noise allowed in decoded content. 0. <= noise <= 1.",
171 )
172 parser.add_argument(
173 "--version",
174 action="version",
175 version="Charset-Normalizer {} - Python {} - Unicode {} - SpeedUp {}".format(
176 __version__,
177 python_version(),
178 unidata_version,
179 "OFF" if md_module.__file__.lower().endswith(".py") else "ON",
180 ),
181 help="Show version information and exit.",
182 )
183
184 args = parser.parse_args(argv)
185
186 if args.replace is True and args.normalize is False:
187 if args.files:
188 for my_file in args.files:
189 my_file.close()
190 print("Use --replace in addition of --normalize only.", file=sys.stderr)
191 return 1
192
193 if args.force is True and args.replace is False:
194 if args.files:
195 for my_file in args.files:
196 my_file.close()
197 print("Use --force in addition of --replace only.", file=sys.stderr)
198 return 1
199
200 if args.threshold < 0.0 or args.threshold > 1.0:
201 if args.files:
202 for my_file in args.files:
203 my_file.close()
204 print("--threshold VALUE should be between 0. AND 1.", file=sys.stderr)
205 return 1
206
207 x_ = []
208
209 for my_file in args.files:
210 matches = from_fp(
211 my_file,
212 threshold=args.threshold,
213 explain=args.verbose,
214 preemptive_behaviour=args.no_preemptive is False,
215 )
216
217 best_guess = matches.best()
218
219 if best_guess is None:
220 print(
221 'Unable to identify originating encoding for "{}". {}'.format(
222 my_file.name,
223 (
224 "Maybe try increasing maximum amount of chaos."
225 if args.threshold < 1.0
226 else ""
227 ),
228 ),
229 file=sys.stderr,
230 )
231 x_.append(
232 CliDetectionResult(
233 abspath(my_file.name),
234 None,
235 [],
236 [],
237 "Unknown",
238 [],
239 False,
240 1.0,
241 0.0,
242 None,
243 True,
244 )
245 )
246 else:
247 cli_result = CliDetectionResult(
248 abspath(my_file.name),
249 best_guess.encoding,
250 best_guess.encoding_aliases,
251 [
252 cp
253 for cp in best_guess.could_be_from_charset
254 if cp != best_guess.encoding
255 ],
256 best_guess.language,
257 best_guess.alphabets,
258 best_guess.bom,
259 best_guess.percent_chaos,
260 best_guess.percent_coherence,
261 None,
262 True,
263 )
264 x_.append(cli_result)
265
266 if len(matches) > 1 and args.alternatives:
267 for el in matches:
268 if el != best_guess:
269 x_.append(
270 CliDetectionResult(
271 abspath(my_file.name),
272 el.encoding,
273 el.encoding_aliases,
274 [
275 cp
276 for cp in el.could_be_from_charset
277 if cp != el.encoding
278 ],
279 el.language,
280 el.alphabets,
281 el.bom,
282 el.percent_chaos,
283 el.percent_coherence,
284 None,
285 False,
286 )
287 )
288
289 if args.normalize is True:
290 if best_guess.encoding.startswith("utf") is True:
291 print(
292 '"{}" file does not need to be normalized, as it already came from unicode.'.format(
293 my_file.name
294 ),
295 file=sys.stderr,
296 )
297 if my_file.closed is False:
298 my_file.close()
299 continue
300
301 dir_path = dirname(realpath(my_file.name))
302 file_name = basename(realpath(my_file.name))
303
304 o_: list[str] = file_name.split(".")
305
306 if args.replace is False:
307 o_.insert(-1, best_guess.encoding)
308 if my_file.closed is False:
309 my_file.close()
310 elif (
311 args.force is False
312 and query_yes_no(
313 'Are you sure to normalize "{}" by replacing it ?'.format(
314 my_file.name
315 ),
316 "no",
317 )
318 is False
319 ):
320 if my_file.closed is False:
321 my_file.close()
322 continue
323
324 try:
325 cli_result.unicode_path = join(dir_path, ".".join(o_))
326
327 with open(cli_result.unicode_path, "wb") as fp:
328 fp.write(best_guess.output())
329 except OSError as e: # Defensive:
330 print(str(e), file=sys.stderr)
331 if my_file.closed is False:
332 my_file.close()
333 return 2
334
335 if my_file.closed is False:
336 my_file.close()
337
338 if args.minimal is False:
339 print(
340 dumps(
341 [el.__dict__ for el in x_] if len(x_) > 1 else x_[0].__dict__,
342 ensure_ascii=True,
343 indent=4,
344 )
345 )
346 else:
347 for my_file in args.files:
348 print(
349 ", ".join(
350 [
351 el.encoding or "undefined"
352 for el in x_
353 if el.path == abspath(my_file.name)
354 ]
355 )
356 )
357
358 return 0
359
360
361if __name__ == "__main__": # Defensive:
362 cli_detect()
363 