Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
__main__.py363 linesDownload Raw Back to cli
1from __future__ import annotations
2
3import argparse
4import sys
5import typing
6from json import dumps
7from os.path import abspath, basename, dirname, join, realpath
8from platform import python_version
9from unicodedata import unidata_version
10
11import charset_normalizer.md as md_module
12from charset_normalizer import from_fp
13from charset_normalizer.models import CliDetectionResult
14from charset_normalizer.version import __version__
15
16
17def query_yes_no(question: str, default: str = "yes") -> bool:  # Defensive:
18    """Ask a yes/no question via input() and return the answer as a bool."""
19    prompt = " [Y/n] " if default == "yes" else " [y/N] "
20
21    while True:
22        choice = input(question + prompt).strip().lower()
23        if not choice:
24            return default == "yes"
25        if choice in ("y", "yes"):
26            return True
27        if choice in ("n", "no"):
28            return False
29        print("Please respond with 'y' or 'n'.")
30
31
32class FileType:
33    """Factory for creating file object types
34
35    Instances of FileType are typically passed as type= arguments to the
36    ArgumentParser add_argument() method.
37
38    Keyword Arguments:
39        - mode -- A string indicating how the file is to be opened. Accepts the
40            same values as the builtin open() function.
41        - bufsize -- The file's desired buffer size. Accepts the same values as
42            the builtin open() function.
43        - encoding -- The file's encoding. Accepts the same values as the
44            builtin open() function.
45        - errors -- A string indicating how encoding and decoding errors are to
46            be handled. Accepts the same value as the builtin open() function.
47
48    Backported from CPython 3.12
49    """
50
51    def __init__(
52        self,
53        mode: str = "r",
54        bufsize: int = -1,
55        encoding: str | None = None,
56        errors: str | None = None,
57    ):
58        self._mode = mode
59        self._bufsize = bufsize
60        self._encoding = encoding
61        self._errors = errors
62
63    def __call__(self, string: str) -> typing.IO:  # type: ignore[type-arg]
64        # the special argument "-" means sys.std{in,out}
65        if string == "-":
66            if "r" in self._mode:
67                return sys.stdin.buffer if "b" in self._mode else sys.stdin
68            elif any(c in self._mode for c in "wax"):
69                return sys.stdout.buffer if "b" in self._mode else sys.stdout
70            else:
71                msg = f'argument "-" with mode {self._mode}'
72                raise ValueError(msg)
73
74        # all other arguments are used as file names
75        try:
76            return open(string, self._mode, self._bufsize, self._encoding, self._errors)
77        except OSError as e:
78            message = f"can't open '{string}': {e}"
79            raise argparse.ArgumentTypeError(message)
80
81    def __repr__(self) -> str:
82        args = self._mode, self._bufsize
83        kwargs = [("encoding", self._encoding), ("errors", self._errors)]
84        args_str = ", ".join(
85            [repr(arg) for arg in args if arg != -1]
86            + [f"{kw}={arg!r}" for kw, arg in kwargs if arg is not None]
87        )
88        return f"{type(self).__name__}({args_str})"
89
90
91def cli_detect(argv: list[str] | None = None) -> int:
92    """
93    CLI assistant using ARGV and ArgumentParser
94    :param argv:
95    :return: 0 if everything is fine, anything else equal trouble
96    """
97    parser = argparse.ArgumentParser(
98        description="The Real First Universal Charset Detector. "
99        "Discover originating encoding used on text file. "
100        "Normalize text to unicode."
101    )
102
103    parser.add_argument(
104        "files", type=FileType("rb"), nargs="+", help="File(s) to be analysed"
105    )
106    parser.add_argument(
107        "-v",
108        "--verbose",
109        action="store_true",
110        default=False,
111        dest="verbose",
112        help="Display complementary information about file if any. "
113        "Stdout will contain logs about the detection process.",
114    )
115    parser.add_argument(
116        "-a",
117        "--with-alternative",
118        action="store_true",
119        default=False,
120        dest="alternatives",
121        help="Output complementary possibilities if any. Top-level JSON WILL be a list.",
122    )
123    parser.add_argument(
124        "-n",
125        "--normalize",
126        action="store_true",
127        default=False,
128        dest="normalize",
129        help="Permit to normalize input file. If not set, program does not write anything.",
130    )
131    parser.add_argument(
132        "-m",
133        "--minimal",
134        action="store_true",
135        default=False,
136        dest="minimal",
137        help="Only output the charset detected to STDOUT. Disabling JSON output.",
138    )
139    parser.add_argument(
140        "-r",
141        "--replace",
142        action="store_true",
143        default=False,
144        dest="replace",
145        help="Replace file when trying to normalize it instead of creating a new one.",
146    )
147    parser.add_argument(
148        "-f",
149        "--force",
150        action="store_true",
151        default=False,
152        dest="force",
153        help="Replace file without asking if you are sure, use this flag with caution.",
154    )
155    parser.add_argument(
156        "-i",
157        "--no-preemptive",
158        action="store_true",
159        default=False,
160        dest="no_preemptive",
161        help="Disable looking at a charset declaration to hint the detector.",
162    )
163    parser.add_argument(
164        "-t",
165        "--threshold",
166        action="store",
167        default=0.2,
168        type=float,
169        dest="threshold",
170        help="Define a custom maximum amount of noise allowed in decoded content. 0. <= noise <= 1.",
171    )
172    parser.add_argument(
173        "--version",
174        action="version",
175        version="Charset-Normalizer {} - Python {} - Unicode {} - SpeedUp {}".format(
176            __version__,
177            python_version(),
178            unidata_version,
179            "OFF" if md_module.__file__.lower().endswith(".py") else "ON",
180        ),
181        help="Show version information and exit.",
182    )
183
184    args = parser.parse_args(argv)
185
186    if args.replace is True and args.normalize is False:
187        if args.files:
188            for my_file in args.files:
189                my_file.close()
190        print("Use --replace in addition of --normalize only.", file=sys.stderr)
191        return 1
192
193    if args.force is True and args.replace is False:
194        if args.files:
195            for my_file in args.files:
196                my_file.close()
197        print("Use --force in addition of --replace only.", file=sys.stderr)
198        return 1
199
200    if args.threshold < 0.0 or args.threshold > 1.0:
201        if args.files:
202            for my_file in args.files:
203                my_file.close()
204        print("--threshold VALUE should be between 0. AND 1.", file=sys.stderr)
205        return 1
206
207    x_ = []
208
209    for my_file in args.files:
210        matches = from_fp(
211            my_file,
212            threshold=args.threshold,
213            explain=args.verbose,
214            preemptive_behaviour=args.no_preemptive is False,
215        )
216
217        best_guess = matches.best()
218
219        if best_guess is None:
220            print(
221                'Unable to identify originating encoding for "{}". {}'.format(
222                    my_file.name,
223                    (
224                        "Maybe try increasing maximum amount of chaos."
225                        if args.threshold < 1.0
226                        else ""
227                    ),
228                ),
229                file=sys.stderr,
230            )
231            x_.append(
232                CliDetectionResult(
233                    abspath(my_file.name),
234                    None,
235                    [],
236                    [],
237                    "Unknown",
238                    [],
239                    False,
240                    1.0,
241                    0.0,
242                    None,
243                    True,
244                )
245            )
246        else:
247            cli_result = CliDetectionResult(
248                abspath(my_file.name),
249                best_guess.encoding,
250                best_guess.encoding_aliases,
251                [
252                    cp
253                    for cp in best_guess.could_be_from_charset
254                    if cp != best_guess.encoding
255                ],
256                best_guess.language,
257                best_guess.alphabets,
258                best_guess.bom,
259                best_guess.percent_chaos,
260                best_guess.percent_coherence,
261                None,
262                True,
263            )
264            x_.append(cli_result)
265
266            if len(matches) > 1 and args.alternatives:
267                for el in matches:
268                    if el != best_guess:
269                        x_.append(
270                            CliDetectionResult(
271                                abspath(my_file.name),
272                                el.encoding,
273                                el.encoding_aliases,
274                                [
275                                    cp
276                                    for cp in el.could_be_from_charset
277                                    if cp != el.encoding
278                                ],
279                                el.language,
280                                el.alphabets,
281                                el.bom,
282                                el.percent_chaos,
283                                el.percent_coherence,
284                                None,
285                                False,
286                            )
287                        )
288
289            if args.normalize is True:
290                if best_guess.encoding.startswith("utf") is True:
291                    print(
292                        '"{}" file does not need to be normalized, as it already came from unicode.'.format(
293                            my_file.name
294                        ),
295                        file=sys.stderr,
296                    )
297                    if my_file.closed is False:
298                        my_file.close()
299                    continue
300
301                dir_path = dirname(realpath(my_file.name))
302                file_name = basename(realpath(my_file.name))
303
304                o_: list[str] = file_name.split(".")
305
306                if args.replace is False:
307                    o_.insert(-1, best_guess.encoding)
308                    if my_file.closed is False:
309                        my_file.close()
310                elif (
311                    args.force is False
312                    and query_yes_no(
313                        'Are you sure to normalize "{}" by replacing it ?'.format(
314                            my_file.name
315                        ),
316                        "no",
317                    )
318                    is False
319                ):
320                    if my_file.closed is False:
321                        my_file.close()
322                    continue
323
324                try:
325                    cli_result.unicode_path = join(dir_path, ".".join(o_))
326
327                    with open(cli_result.unicode_path, "wb") as fp:
328                        fp.write(best_guess.output())
329                except OSError as e:  # Defensive:
330                    print(str(e), file=sys.stderr)
331                    if my_file.closed is False:
332                        my_file.close()
333                    return 2
334
335        if my_file.closed is False:
336            my_file.close()
337
338    if args.minimal is False:
339        print(
340            dumps(
341                [el.__dict__ for el in x_] if len(x_) > 1 else x_[0].__dict__,
342                ensure_ascii=True,
343                indent=4,
344            )
345        )
346    else:
347        for my_file in args.files:
348            print(
349                ", ".join(
350                    [
351                        el.encoding or "undefined"
352                        for el in x_
353                        if el.path == abspath(my_file.name)
354                    ]
355                )
356            )
357
358    return 0
359
360
361if __name__ == "__main__":  # Defensive:
362    cli_detect()
363 
codekingpro/portable-devtools · Team Ai