Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
httputil_test.py773 linesDownload Raw Back to test
1from tornado.httputil import (
2    url_concat,
3    parse_multipart_form_data,
4    HTTPHeaders,
5    format_timestamp,
6    HTTPServerRequest,
7    parse_request_start_line,
8    parse_cookie,
9    qs_to_qsl,
10    HTTPInputError,
11    HTTPFile,
12    ParseMultipartConfig,
13)
14from tornado.escape import utf8, native_str
15from tornado.log import gen_log
16from tornado.test.util import ignore_deprecation
17
18import copy
19import datetime
20import logging
21import pickle
22import time
23import urllib.parse
24import unittest
25
26from typing import Tuple, Dict, List
27
28
29def form_data_args() -> Tuple[Dict[str, List[bytes]], Dict[str, List[HTTPFile]]]:
30    """Return two empty dicts suitable for use with parse_multipart_form_data.
31
32    mypy insists on type annotations for dict literals, so this lets us avoid
33    the verbose types throughout this test.
34    """
35    return {}, {}
36
37
38class TestUrlConcat(unittest.TestCase):
39    def test_url_concat_no_query_params(self):
40        url = url_concat("https://localhost/path", [("y", "y"), ("z", "z")])
41        self.assertEqual(url, "https://localhost/path?y=y&z=z")
42
43    def test_url_concat_encode_args(self):
44        url = url_concat("https://localhost/path", [("y", "/y"), ("z", "z")])
45        self.assertEqual(url, "https://localhost/path?y=%2Fy&z=z")
46
47    def test_url_concat_trailing_q(self):
48        url = url_concat("https://localhost/path?", [("y", "y"), ("z", "z")])
49        self.assertEqual(url, "https://localhost/path?y=y&z=z")
50
51    def test_url_concat_q_with_no_trailing_amp(self):
52        url = url_concat("https://localhost/path?x", [("y", "y"), ("z", "z")])
53        self.assertEqual(url, "https://localhost/path?x=&y=y&z=z")
54
55    def test_url_concat_trailing_amp(self):
56        url = url_concat("https://localhost/path?x&", [("y", "y"), ("z", "z")])
57        self.assertEqual(url, "https://localhost/path?x=&y=y&z=z")
58
59    def test_url_concat_mult_params(self):
60        url = url_concat("https://localhost/path?a=1&b=2", [("y", "y"), ("z", "z")])
61        self.assertEqual(url, "https://localhost/path?a=1&b=2&y=y&z=z")
62
63    def test_url_concat_no_params(self):
64        url = url_concat("https://localhost/path?r=1&t=2", [])
65        self.assertEqual(url, "https://localhost/path?r=1&t=2")
66
67    def test_url_concat_none_params(self):
68        url = url_concat("https://localhost/path?r=1&t=2", None)
69        self.assertEqual(url, "https://localhost/path?r=1&t=2")
70
71    def test_url_concat_with_frag(self):
72        url = url_concat("https://localhost/path#tab", [("y", "y")])
73        self.assertEqual(url, "https://localhost/path?y=y#tab")
74
75    def test_url_concat_multi_same_params(self):
76        url = url_concat("https://localhost/path", [("y", "y1"), ("y", "y2")])
77        self.assertEqual(url, "https://localhost/path?y=y1&y=y2")
78
79    def test_url_concat_multi_same_query_params(self):
80        url = url_concat("https://localhost/path?r=1&r=2", [("y", "y")])
81        self.assertEqual(url, "https://localhost/path?r=1&r=2&y=y")
82
83    def test_url_concat_dict_params(self):
84        url = url_concat("https://localhost/path", dict(y="y"))
85        self.assertEqual(url, "https://localhost/path?y=y")
86
87
88class QsParseTest(unittest.TestCase):
89    def test_parsing(self):
90        qsstring = "a=1&b=2&a=3"
91        qs = urllib.parse.parse_qs(qsstring)
92        qsl = list(qs_to_qsl(qs))
93        self.assertIn(("a", "1"), qsl)
94        self.assertIn(("a", "3"), qsl)
95        self.assertIn(("b", "2"), qsl)
96
97
98class MultipartFormDataTest(unittest.TestCase):
99    def test_file_upload(self):
100        data = b"""\
101--1234
102Content-Disposition: form-data; name="files"; filename="ab.txt"
103
104Foo
105--1234--""".replace(
106            b"\n", b"\r\n"
107        )
108        args, files = form_data_args()
109        parse_multipart_form_data(b"1234", data, args, files)
110        file = files["files"][0]
111        self.assertEqual(file["filename"], "ab.txt")
112        self.assertEqual(file["body"], b"Foo")
113
114    def test_unquoted_names(self):
115        # quotes are optional unless special characters are present
116        data = b"""\
117--1234
118Content-Disposition: form-data; name=files; filename=ab.txt
119
120Foo
121--1234--""".replace(
122            b"\n", b"\r\n"
123        )
124        args, files = form_data_args()
125        parse_multipart_form_data(b"1234", data, args, files)
126        file = files["files"][0]
127        self.assertEqual(file["filename"], "ab.txt")
128        self.assertEqual(file["body"], b"Foo")
129
130    def test_special_filenames(self):
131        filenames = [
132            "a;b.txt",
133            'a"b.txt',
134            'a";b.txt',
135            'a;"b.txt',
136            'a";";.txt',
137            'a\\"b.txt',
138            "a\\b.txt",
139            "a b.txt",
140            "a\tb.txt",
141        ]
142        for filename in filenames:
143            logging.debug("trying filename %r", filename)
144            str_data = """\
145--1234
146Content-Disposition: form-data; name="files"; filename="%s"
147
148Foo
149--1234--""" % filename.replace(
150                "\\", "\\\\"
151            ).replace(
152                '"', '\\"'
153            )
154            data = utf8(str_data.replace("\n", "\r\n"))
155            args, files = form_data_args()
156            parse_multipart_form_data(b"1234", data, args, files)
157            file = files["files"][0]
158            self.assertEqual(file["filename"], filename)
159            self.assertEqual(file["body"], b"Foo")
160
161    def test_invalid_chars(self):
162        filenames = [
163            "a\rb.txt",
164            "a\0b.txt",
165            "a\x08b.txt",
166        ]
167        for filename in filenames:
168            str_data = """\
169--1234
170Content-Disposition: form-data; name="files"; filename="%s"
171
172Foo
173--1234--""" % filename.replace(
174                "\\", "\\\\"
175            ).replace(
176                '"', '\\"'
177            )
178            data = utf8(str_data.replace("\n", "\r\n"))
179            args, files = form_data_args()
180            with self.assertRaises(HTTPInputError) as cm:
181                parse_multipart_form_data(b"1234", data, args, files)
182            self.assertIn("Invalid header value", str(cm.exception))
183
184    def test_non_ascii_filename_rfc5987(self):
185        data = b"""\
186--1234
187Content-Disposition: form-data; name="files"; filename="ab.txt"; filename*=UTF-8''%C3%A1b.txt
188
189Foo
190--1234--""".replace(
191            b"\n", b"\r\n"
192        )
193        args, files = form_data_args()
194        parse_multipart_form_data(b"1234", data, args, files)
195        file = files["files"][0]
196        self.assertEqual(file["filename"], "áb.txt")
197        self.assertEqual(file["body"], b"Foo")
198
199    def test_non_ascii_filename_raw(self):
200        data = """\
201--1234
202Content-Disposition: form-data; name="files"; filename="测试.txt"
203
204Foo
205--1234--""".encode(
206            "utf-8"
207        ).replace(
208            b"\n", b"\r\n"
209        )
210        args, files = form_data_args()
211        parse_multipart_form_data(b"1234", data, args, files)
212        file = files["files"][0]
213        self.assertEqual(file["filename"], "测试.txt")
214        self.assertEqual(file["body"], b"Foo")
215
216    def test_boundary_starts_and_ends_with_quotes(self):
217        data = b"""\
218--1234
219Content-Disposition: form-data; name="files"; filename="ab.txt"
220
221Foo
222--1234--""".replace(
223            b"\n", b"\r\n"
224        )
225        args, files = form_data_args()
226        parse_multipart_form_data(b'"1234"', data, args, files)
227        file = files["files"][0]
228        self.assertEqual(file["filename"], "ab.txt")
229        self.assertEqual(file["body"], b"Foo")
230
231    def test_missing_headers(self):
232        data = b"""\
233--1234
234
235Foo
236--1234--""".replace(
237            b"\n", b"\r\n"
238        )
239        args, files = form_data_args()
240        with self.assertRaises(
241            HTTPInputError, msg="multipart/form-data missing headers"
242        ):
243            parse_multipart_form_data(b"1234", data, args, files)
244        self.assertEqual(files, {})
245
246    def test_invalid_content_disposition(self):
247        data = b"""\
248--1234
249Content-Disposition: invalid; name="files"; filename="ab.txt"
250
251Foo
252--1234--""".replace(
253            b"\n", b"\r\n"
254        )
255        args, files = form_data_args()
256        with self.assertRaises(HTTPInputError, msg="Invalid multipart/form-data"):
257            parse_multipart_form_data(b"1234", data, args, files)
258        self.assertEqual(files, {})
259
260    def test_line_does_not_end_with_correct_line_break(self):
261        data = b"""\
262--1234
263Content-Disposition: form-data; name="files"; filename="ab.txt"
264
265Foo--1234--""".replace(
266            b"\n", b"\r\n"
267        )
268        args, files = form_data_args()
269        with self.assertRaises(HTTPInputError, msg="Invalid multipart/form-data"):
270            parse_multipart_form_data(b"1234", data, args, files)
271        self.assertEqual(files, {})
272
273    def test_content_disposition_header_without_name_parameter(self):
274        data = b"""\
275--1234
276Content-Disposition: form-data; filename="ab.txt"
277
278Foo
279--1234--""".replace(
280            b"\n", b"\r\n"
281        )
282        args, files = form_data_args()
283        with self.assertRaises(
284            HTTPInputError, msg="multipart/form-data value missing name"
285        ):
286            parse_multipart_form_data(b"1234", data, args, files)
287        self.assertEqual(files, {})
288
289    def test_data_after_final_boundary(self):
290        # The spec requires that data after the final boundary be ignored.
291        # http://www.w3.org/Protocols/rfc1341/7_2_Multipart.html
292        # In practice, some libraries include an extra CRLF after the boundary.
293        data = b"""\
294--1234
295Content-Disposition: form-data; name="files"; filename="ab.txt"
296
297Foo
298--1234--
299""".replace(
300            b"\n", b"\r\n"
301        )
302        args, files = form_data_args()
303        parse_multipart_form_data(b"1234", data, args, files)
304        file = files["files"][0]
305        self.assertEqual(file["filename"], "ab.txt")
306        self.assertEqual(file["body"], b"Foo")
307
308    def test_disposition_param_linear_performance(self):
309        # This is a regression test for performance of parsing parameters
310        # to the content-disposition header, specifically for semicolons within
311        # quoted strings.
312        def f(n):
313            start = time.perf_counter()
314            message = (
315                b"--1234\r\nContent-Disposition: form-data; "
316                + b'x="'
317                + b";" * n
318                + b'"; '
319                + b'name="files"; filename="a.txt"\r\n\r\nFoo\r\n--1234--\r\n'
320            )
321            args: dict[str, list[bytes]] = {}
322            files: dict[str, list[HTTPFile]] = {}
323            parse_multipart_form_data(b"1234", message, args, files)
324            return time.perf_counter() - start
325
326        d1 = f(1_000)
327        # Note that headers larger than this are blocked by the default configuration.
328        d2 = f(10_000)
329        if d2 / d1 > 20:
330            self.fail(f"Disposition param parsing is not linear: {d1=} vs {d2=}")
331
332    def test_multipart_config(self):
333        boundary = b"1234"
334        body = b"""--1234
335Content-Disposition: form-data; name="files"; filename="ab.txt"
336
337--1234--""".replace(
338            b"\n", b"\r\n"
339        )
340        config = ParseMultipartConfig()
341        args, files = form_data_args()
342        parse_multipart_form_data(boundary, body, args, files, config=config)
343        self.assertEqual(files["files"][0]["filename"], "ab.txt")
344
345        config_no_parts = ParseMultipartConfig(max_parts=0)
346        with self.assertRaises(HTTPInputError) as cm:
347            parse_multipart_form_data(
348                boundary, body, args, files, config=config_no_parts
349            )
350        self.assertIn("too many parts", str(cm.exception))
351
352        config_small_headers = ParseMultipartConfig(max_part_header_size=10)
353        with self.assertRaises(HTTPInputError) as cm:
354            parse_multipart_form_data(
355                boundary, body, args, files, config=config_small_headers
356            )
357        self.assertIn("header too large", str(cm.exception))
358
359        config_disabled = ParseMultipartConfig(enabled=False)
360        with self.assertRaises(HTTPInputError) as cm:
361            parse_multipart_form_data(
362                boundary, body, args, files, config=config_disabled
363            )
364        self.assertIn("multipart/form-data parsing is disabled", str(cm.exception))
365
366
367class HTTPHeadersTest(unittest.TestCase):
368    def test_multi_line(self):
369        # Lines beginning with whitespace are appended to the previous line
370        # with any leading whitespace replaced by a single space.
371        # Note that while multi-line headers are a part of the HTTP spec,
372        # their use is strongly discouraged.
373        data = """\
374Foo: bar
375 baz
376Asdf: qwer
377\tzxcv
378Foo: even
379     more
380     lines
381""".replace(
382            "\n", "\r\n"
383        )
384        headers = HTTPHeaders.parse(data)
385        self.assertEqual(headers["asdf"], "qwer zxcv")
386        self.assertEqual(headers.get_list("asdf"), ["qwer zxcv"])
387        self.assertEqual(headers["Foo"], "bar baz,even more lines")
388        self.assertEqual(headers.get_list("foo"), ["bar baz", "even more lines"])
389        self.assertEqual(
390            sorted(list(headers.get_all())),
391            [("Asdf", "qwer zxcv"), ("Foo", "bar baz"), ("Foo", "even more lines")],
392        )
393        # Verify case insensitivity in-operator
394        self.assertTrue("asdf" in headers)
395        self.assertTrue("Asdf" in headers)
396
397    def test_continuation(self):
398        data = "Foo: bar\r\n\tasdf"
399        headers = HTTPHeaders.parse(data)
400        self.assertEqual(headers["Foo"], "bar asdf")
401
402        # If the first line starts with whitespace, it's a
403        # continuation line with nothing to continue, so reject it
404        # (with a proper error).
405        data = " Foo: bar"
406        self.assertRaises(HTTPInputError, HTTPHeaders.parse, data)
407
408        # \f (formfeed) is whitespace according to str.isspace, but
409        # not according to the HTTP spec.
410        data = "Foo: bar\r\n\fasdf"
411        self.assertRaises(HTTPInputError, HTTPHeaders.parse, data)
412
413    def test_forbidden_ascii_characters(self):
414        # Control characters and ASCII whitespace other than space, tab, and CRLF are not allowed in
415        # headers.
416        for c in range(0xFF):
417            data = f"Foo: bar{chr(c)}baz\r\n"
418            if c == 0x09 or (c >= 0x20 and c != 0x7F):
419                headers = HTTPHeaders.parse(data)
420                self.assertEqual(headers["Foo"], f"bar{chr(c)}baz")
421            else:
422                self.assertRaises(HTTPInputError, HTTPHeaders.parse, data)
423
424    def test_unicode_newlines(self):
425        # Ensure that only \r\n is recognized as a header separator, and not
426        # the other newline-like unicode characters.
427        # Characters that are likely to be problematic can be found in
428        # http://unicode.org/standard/reports/tr13/tr13-5.html
429        # and cpython's unicodeobject.c (which defines the implementation
430        # of unicode_type.splitlines(), and uses a different list than TR13).
431        newlines = [
432            # The following ascii characters are sometimes treated as newline-like,
433            # but they're disallowed in HTTP headers. This test covers unicode
434            # characters that are permitted in headers (under the obs-text rule).
435            # "\u001b",  # VERTICAL TAB
436            # "\u001c",  # FILE SEPARATOR
437            # "\u001d",  # GROUP SEPARATOR
438            # "\u001e",  # RECORD SEPARATOR
439            "\u0085",  # NEXT LINE
440            "\u2028",  # LINE SEPARATOR
441            "\u2029",  # PARAGRAPH SEPARATOR
442        ]
443        for newline in newlines:
444            # Try the utf8 and latin1 representations of each newline
445            for encoding in ["utf8", "latin1"]:
446                try:
447                    try:
448                        encoded = newline.encode(encoding)
449                    except UnicodeEncodeError:
450                        # Some chars cannot be represented in latin1
451                        continue
452                    data = b"Cookie: foo=" + encoded + b"bar"
453                    # parse() wants a native_str, so decode through latin1
454                    # in the same way the real parser does.
455                    headers = HTTPHeaders.parse(native_str(data.decode("latin1")))
456                    expected = [
457                        (
458                            "Cookie",
459                            "foo=" + native_str(encoded.decode("latin1")) + "bar",
460                        )
461                    ]
462                    self.assertEqual(expected, list(headers.get_all()))
463                except Exception:
464                    gen_log.warning("failed while trying %r in %s", newline, encoding)
465                    raise
466
467    def test_unicode_whitespace(self):
468        # Only tabs and spaces are to be stripped according to the HTTP standard.
469        # Other unicode whitespace is to be left as-is. In the context of headers,
470        # this specifically means the whitespace characters falling within the
471        # latin1 charset.
472        whitespace = [
473            (" ", True),  # SPACE
474            ("\t", True),  # TAB
475            ("\u00a0", False),  # NON-BREAKING SPACE
476            ("\u0085", False),  # NEXT LINE
477        ]
478        for c, stripped in whitespace:
479            headers = HTTPHeaders.parse("Transfer-Encoding: %schunked" % c)
480            if stripped:
481                expected = [("Transfer-Encoding", "chunked")]
482            else:
483                expected = [("Transfer-Encoding", "%schunked" % c)]
484            self.assertEqual(expected, list(headers.get_all()))
485
486    def test_optional_cr(self):
487        # Bare CR is  not a valid line separator
488        with self.assertRaises(HTTPInputError):
489            HTTPHeaders.parse("CRLF: crlf\r\nLF: lf\nCR: cr\rMore: more\r\n")
490
491        # Both CRLF and LF should be accepted as separators. CR should not be
492        # part of the data when followed by LF.
493        headers = HTTPHeaders.parse("CRLF: crlf\r\nLF: lf\nMore: more\r\n")
494        self.assertEqual(
495            sorted(headers.get_all()),
496            [("Crlf", "crlf"), ("Lf", "lf"), ("More", "more")],
497        )
498
499    def test_copy(self):
500        all_pairs = [("A", "1"), ("A", "2"), ("B", "c")]
501        h1 = HTTPHeaders()
502        for k, v in all_pairs:
503            h1.add(k, v)
504        h2 = h1.copy()
505        h3 = copy.copy(h1)
506        h4 = copy.deepcopy(h1)
507        for headers in [h1, h2, h3, h4]:
508            # All the copies are identical, no matter how they were
509            # constructed.
510            self.assertEqual(list(sorted(headers.get_all())), all_pairs)
511        for headers in [h2, h3, h4]:
512            # Neither the dict or its member lists are reused.
513            self.assertIsNot(headers, h1)
514            self.assertIsNot(headers.get_list("A"), h1.get_list("A"))
515
516    def test_pickle_roundtrip(self):
517        headers = HTTPHeaders()
518        headers.add("Set-Cookie", "a=b")
519        headers.add("Set-Cookie", "c=d")
520        headers.add("Content-Type", "text/html")
521        pickled = pickle.dumps(headers)
522        unpickled = pickle.loads(pickled)
523        self.assertEqual(sorted(headers.get_all()), sorted(unpickled.get_all()))
524        self.assertEqual(sorted(headers.items()), sorted(unpickled.items()))
525
526    def test_setdefault(self):
527        headers = HTTPHeaders()
528        headers["foo"] = "bar"
529        # If a value is present, setdefault returns it without changes.
530        self.assertEqual(headers.setdefault("foo", "baz"), "bar")
531        self.assertEqual(headers["foo"], "bar")
532        # If a value is not present, setdefault sets it for future use.
533        self.assertEqual(headers.setdefault("quux", "xyzzy"), "xyzzy")
534        self.assertEqual(headers["quux"], "xyzzy")
535        self.assertEqual(sorted(headers.get_all()), [("Foo", "bar"), ("Quux", "xyzzy")])
536
537    def test_string(self):
538        headers = HTTPHeaders()
539        headers.add("Foo", "1")
540        headers.add("Foo", "2")
541        headers.add("Foo", "3")
542        headers2 = HTTPHeaders.parse(str(headers))
543        self.assertEqual(headers, headers2)
544
545    def test_invalid_header_names(self):
546        invalid_names = [
547            "",
548            "foo bar",
549            "foo\tbar",
550            "foo\nbar",
551            "foo\x00bar",
552            "foo ",
553            " foo",
554            "é",
555        ]
556        for name in invalid_names:
557            headers = HTTPHeaders()
558            with self.assertRaises(HTTPInputError):
559                headers.add(name, "bar")
560
561    def test_linear_performance(self):
562        def f(n):
563            start = time.perf_counter()
564            headers = HTTPHeaders()
565            for i in range(n):
566                headers.add("X-Foo", "bar")
567            return time.perf_counter() - start
568
569        # This runs under 50ms on my laptop as of 2025-12-09.
570        d1 = f(10_000)
571        d2 = f(100_000)
572        if d2 / d1 > 20:
573            # d2 should be about 10x d1 but allow a wide margin for variability.
574            self.fail(f"HTTPHeaders.add() does not scale linearly: {d1=} vs {d2=}")
575
576
577class FormatTimestampTest(unittest.TestCase):
578    # Make sure that all the input types are supported.
579    TIMESTAMP = 1359312200.503611
580    EXPECTED = "Sun, 27 Jan 2013 18:43:20 GMT"
581
582    def check(self, value):
583        self.assertEqual(format_timestamp(value), self.EXPECTED)
584
585    def test_unix_time_float(self):
586        self.check(self.TIMESTAMP)
587
588    def test_unix_time_int(self):
589        self.check(int(self.TIMESTAMP))
590
591    def test_struct_time(self):
592        self.check(time.gmtime(self.TIMESTAMP))
593
594    def test_time_tuple(self):
595        tup = tuple(time.gmtime(self.TIMESTAMP))
596        self.assertEqual(9, len(tup))
597        self.check(tup)
598
599    def test_utc_naive_datetime(self):
600        self.check(
601            datetime.datetime.fromtimestamp(
602                self.TIMESTAMP, datetime.timezone.utc
603            ).replace(tzinfo=None)
604        )
605
606    def test_utc_naive_datetime_deprecated(self):
607        with ignore_deprecation():
608            self.check(datetime.datetime.utcfromtimestamp(self.TIMESTAMP))
609
610    def test_utc_aware_datetime(self):
611        self.check(
612            datetime.datetime.fromtimestamp(self.TIMESTAMP, datetime.timezone.utc)
613        )
614
615    def test_other_aware_datetime(self):
616        # Other timezones are ignored; the timezone is always printed as GMT
617        self.check(
618            datetime.datetime.fromtimestamp(
619                self.TIMESTAMP, datetime.timezone(datetime.timedelta(hours=-4))
620            )
621        )
622
623
624# HTTPServerRequest is mainly tested incidentally to the server itself,
625# but this tests the parts of the class that can be tested in isolation.
626class HTTPServerRequestTest(unittest.TestCase):
627    def test_default_constructor(self):
628        # All parameters are formally optional, but uri is required
629        # (and has been for some time).  This test ensures that no
630        # more required parameters slip in.
631        HTTPServerRequest(uri="/")
632
633    def test_body_is_a_byte_string(self):
634        request = HTTPServerRequest(uri="/")
635        self.assertIsInstance(request.body, bytes)
636
637    def test_repr_does_not_contain_headers(self):
638        request = HTTPServerRequest(
639            uri="/", headers=HTTPHeaders({"Canary": ["Coal Mine"]})
640        )
641        self.assertNotIn("Canary", repr(request))
642
643
644class ParseRequestStartLineTest(unittest.TestCase):
645    METHOD = "GET"
646    PATH = "/foo"
647    VERSION = "HTTP/1.1"
648
649    def test_parse_request_start_line(self):
650        start_line = " ".join([self.METHOD, self.PATH, self.VERSION])
651        parsed_start_line = parse_request_start_line(start_line)
652        self.assertEqual(parsed_start_line.method, self.METHOD)
653        self.assertEqual(parsed_start_line.path, self.PATH)
654        self.assertEqual(parsed_start_line.version, self.VERSION)
655
656
657class ParseCookieTest(unittest.TestCase):
658    # These tests copied from Django:
659    # https://github.com/django/django/pull/6277/commits/da810901ada1cae9fc1f018f879f11a7fb467b28
660    def test_python_cookies(self):
661        """
662        Test cases copied from Python's Lib/test/test_http_cookies.py
663        """
664        self.assertEqual(
665            parse_cookie("chips=ahoy; vienna=finger"),
666            {"chips": "ahoy", "vienna": "finger"},
667        )
668        # Here parse_cookie() differs from Python's cookie parsing in that it
669        # treats all semicolons as delimiters, even within quotes.
670        self.assertEqual(
671            parse_cookie('keebler="E=mc2; L=\\"Loves\\"; fudge=\\012;"'),
672            {"keebler": '"E=mc2', "L": '\\"Loves\\"', "fudge": "\\012", "": '"'},
673        )
674        # Illegal cookies that have an '=' char in an unquoted value.
675        self.assertEqual(parse_cookie("keebler=E=mc2"), {"keebler": "E=mc2"})
676        # Cookies with ':' character in their name.
677        self.assertEqual(
678            parse_cookie("key:term=value:term"), {"key:term": "value:term"}
679        )
680        # Cookies with '[' and ']'.
681        self.assertEqual(
682            parse_cookie("a=b; c=[; d=r; f=h"), {"a": "b", "c": "[", "d": "r", "f": "h"}
683        )
684
685    def test_cookie_edgecases(self):
686        # Cookies that RFC6265 allows.
687        self.assertEqual(
688            parse_cookie("a=b; Domain=example.com"), {"a": "b", "Domain": "example.com"}
689        )
690        # parse_cookie() has historically kept only the last cookie with the
691        # same name.
692        self.assertEqual(parse_cookie("a=b; h=i; a=c"), {"a": "c", "h": "i"})
693
694    def test_invalid_cookies(self):
695        """
696        Cookie strings that go against RFC6265 but browsers will send if set
697        via document.cookie.
698        """
699        # Chunks without an equals sign appear as unnamed values per
700        # https://bugzilla.mozilla.org/show_bug.cgi?id=169091
701        self.assertIn(
702            "django_language",
703            parse_cookie("abc=def; unnamed; django_language=en").keys(),
704        )
705        # Even a double quote may be an unamed value.
706        self.assertEqual(parse_cookie('a=b; "; c=d'), {"a": "b", "": '"', "c": "d"})
707        # Spaces in names and values, and an equals sign in values.
708        self.assertEqual(
709            parse_cookie("a b c=d e = f; gh=i"), {"a b c": "d e = f", "gh": "i"}
710        )
711        # More characters the spec forbids.
712        self.assertEqual(
713            parse_cookie('a   b,c<>@:/[]?{}=d  "  =e,f g'),
714            {"a   b,c<>@:/[]?{}": 'd  "  =e,f g'},
715        )
716        # Unicode characters. The spec only allows ASCII.
717        self.assertEqual(
718            parse_cookie("saint=André Bessette"),
719            {"saint": native_str("André Bessette")},
720        )
721        # Browsers don't send extra whitespace or semicolons in Cookie headers,
722        # but parse_cookie() should parse whitespace the same way
723        # document.cookie parses whitespace.
724        self.assertEqual(
725            parse_cookie("  =  b  ;  ;  =  ;   c  =  ;  "), {"": "b", "c": ""}
726        )
727
728    def test_unquote(self):
729        # Copied from
730        # https://github.com/python/cpython/blob/dc7a2b6522ec7af41282bc34f405bee9b306d611/Lib/test/test_http_cookies.py#L62
731        cases = [
732            (r'a="b=\""', 'b="'),
733            (r'a="b=\\"', "b=\\"),
734            (r'a="b=\="', "b=="),
735            (r'a="b=\n"', "b=n"),
736            (r'a="b=\042"', 'b="'),
737            (r'a="b=\134"', "b=\\"),
738            (r'a="b=\377"', "b=\xff"),
739            (r'a="b=\400"', "b=400"),
740            (r'a="b=\42"', "b=42"),
741            (r'a="b=\\042"', "b=\\042"),
742            (r'a="b=\\134"', "b=\\134"),
743            (r'a="b=\\\""', 'b=\\"'),
744            (r'a="b=\\\042"', 'b=\\"'),
745            (r'a="b=\134\""', 'b=\\"'),
746            (r'a="b=\134\042"', 'b=\\"'),
747        ]
748        for encoded, decoded in cases:
749            with self.subTest(encoded):
750                c = parse_cookie(encoded)
751                self.assertEqual(c["a"], decoded)
752
753    def test_unquote_large(self):
754        # Adapted from
755        # https://github.com/python/cpython/blob/dc7a2b6522ec7af41282bc34f405bee9b306d611/Lib/test/test_http_cookies.py#L87
756        # Modified from that test because we handle semicolons differently from the stdlib.
757        #
758        # This is a performance regression test: prior to improvements in Tornado 6.4.2, this test
759        # would take over a minute with n= 100k. Now it runs in tens of milliseconds.
760        n = 100000
761        for encoded in r"\\", r"\134":
762            with self.subTest(encoded):
763                start = time.time()
764                data = 'a="b=' + encoded * n + '"'
765                value = parse_cookie(data)["a"]
766                end = time.time()
767                self.assertEqual(value[:3], "b=\\")
768                self.assertEqual(value[-3:], "\\\\\\")
769                self.assertEqual(len(value), n + 2)
770
771                # Very loose performance check to avoid false positives
772                self.assertLess(end - start, 1, "Test took too long")
773 
codekingpro/portable-devtools · Team Ai