Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
escape_test.py325 linesDownload Raw Back to test
1import unittest
2
3import tornado
4from tornado.escape import (
5    utf8,
6    xhtml_escape,
7    xhtml_unescape,
8    url_escape,
9    url_unescape,
10    to_unicode,
11    json_decode,
12    json_encode,
13    squeeze,
14    recursive_unicode,
15)
16from tornado.util import unicode_type
17
18from typing import List, Tuple, Union, Dict, Any  # noqa: F401
19
20linkify_tests = [
21    # (input, linkify_kwargs, expected_output)
22    (
23        "hello http://world.com/!",
24        {},
25        'hello <a href="http://world.com/">http://world.com/</a>!',
26    ),
27    (
28        "hello http://world.com/with?param=true&stuff=yes",
29        {},
30        'hello <a href="http://world.com/with?param=true&amp;stuff=yes">http://world.com/with?param=true&amp;stuff=yes</a>',  # noqa: E501
31    ),
32    # an opened paren followed by many chars killed Gruber's regex
33    (
34        "http://url.com/w(aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
35        {},
36        '<a href="http://url.com/w">http://url.com/w</a>(aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa',  # noqa: E501
37    ),
38    # as did too many dots at the end
39    (
40        "http://url.com/withmany.......................................",
41        {},
42        '<a href="http://url.com/withmany">http://url.com/withmany</a>.......................................',  # noqa: E501
43    ),
44    (
45        "http://url.com/withmany((((((((((((((((((((((((((((((((((a)",
46        {},
47        '<a href="http://url.com/withmany">http://url.com/withmany</a>((((((((((((((((((((((((((((((((((a)',  # noqa: E501
48    ),
49    # some examples from http://daringfireball.net/2009/11/liberal_regex_for_matching_urls
50    # plus a fex extras (such as multiple parentheses).
51    (
52        "http://foo.com/blah_blah",
53        {},
54        '<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>',
55    ),
56    (
57        "http://foo.com/blah_blah/",
58        {},
59        '<a href="http://foo.com/blah_blah/">http://foo.com/blah_blah/</a>',
60    ),
61    (
62        "(Something like http://foo.com/blah_blah)",
63        {},
64        '(Something like <a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>)',
65    ),
66    (
67        "http://foo.com/blah_blah_(wikipedia)",
68        {},
69        '<a href="http://foo.com/blah_blah_(wikipedia)">http://foo.com/blah_blah_(wikipedia)</a>',
70    ),
71    (
72        "http://foo.com/blah_(blah)_(wikipedia)_blah",
73        {},
74        '<a href="http://foo.com/blah_(blah)_(wikipedia)_blah">http://foo.com/blah_(blah)_(wikipedia)_blah</a>',  # noqa: E501
75    ),
76    (
77        "(Something like http://foo.com/blah_blah_(wikipedia))",
78        {},
79        '(Something like <a href="http://foo.com/blah_blah_(wikipedia)">http://foo.com/blah_blah_(wikipedia)</a>)',  # noqa: E501
80    ),
81    (
82        "http://foo.com/blah_blah.",
83        {},
84        '<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>.',
85    ),
86    (
87        "http://foo.com/blah_blah/.",
88        {},
89        '<a href="http://foo.com/blah_blah/">http://foo.com/blah_blah/</a>.',
90    ),
91    (
92        "<http://foo.com/blah_blah>",
93        {},
94        '&lt;<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>&gt;',
95    ),
96    (
97        "<http://foo.com/blah_blah/>",
98        {},
99        '&lt;<a href="http://foo.com/blah_blah/">http://foo.com/blah_blah/</a>&gt;',
100    ),
101    (
102        "http://foo.com/blah_blah,",
103        {},
104        '<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>,',
105    ),
106    (
107        "http://www.example.com/wpstyle/?p=364.",
108        {},
109        '<a href="http://www.example.com/wpstyle/?p=364">http://www.example.com/wpstyle/?p=364</a>.',  # noqa: E501
110    ),
111    (
112        "rdar://1234",
113        {"permitted_protocols": ["http", "rdar"]},
114        '<a href="rdar://1234">rdar://1234</a>',
115    ),
116    (
117        "rdar:/1234",
118        {"permitted_protocols": ["rdar"]},
119        '<a href="rdar:/1234">rdar:/1234</a>',
120    ),
121    (
122        "http://userid:password@example.com:8080",
123        {},
124        '<a href="http://userid:password@example.com:8080">http://userid:password@example.com:8080</a>',  # noqa: E501
125    ),
126    (
127        "http://userid@example.com",
128        {},
129        '<a href="http://userid@example.com">http://userid@example.com</a>',
130    ),
131    (
132        "http://userid@example.com:8080",
133        {},
134        '<a href="http://userid@example.com:8080">http://userid@example.com:8080</a>',
135    ),
136    (
137        "http://userid:password@example.com",
138        {},
139        '<a href="http://userid:password@example.com">http://userid:password@example.com</a>',
140    ),
141    (
142        "message://%3c330e7f8409726r6a4ba78dkf1fd71420c1bf6ff@mail.gmail.com%3e",
143        {"permitted_protocols": ["http", "message"]},
144        '<a href="message://%3c330e7f8409726r6a4ba78dkf1fd71420c1bf6ff@mail.gmail.com%3e">'
145        "message://%3c330e7f8409726r6a4ba78dkf1fd71420c1bf6ff@mail.gmail.com%3e</a>",
146    ),
147    (
148        "http://\u27a1.ws/\u4a39",
149        {},
150        '<a href="http://\u27a1.ws/\u4a39">http://\u27a1.ws/\u4a39</a>',
151    ),
152    (
153        "<tag>http://example.com</tag>",
154        {},
155        '&lt;tag&gt;<a href="http://example.com">http://example.com</a>&lt;/tag&gt;',
156    ),
157    (
158        "Just a www.example.com link.",
159        {},
160        'Just a <a href="http://www.example.com">www.example.com</a> link.',
161    ),
162    (
163        "Just a www.example.com link.",
164        {"require_protocol": True},
165        "Just a www.example.com link.",
166    ),
167    (
168        "A http://reallylong.com/link/that/exceedsthelenglimit.html",
169        {"require_protocol": True, "shorten": True},
170        'A <a href="http://reallylong.com/link/that/exceedsthelenglimit.html"'
171        ' title="http://reallylong.com/link/that/exceedsthelenglimit.html">http://reallylong.com/link...</a>',  # noqa: E501
172    ),
173    (
174        "A http://reallylongdomainnamethatwillbetoolong.com/hi!",
175        {"shorten": True},
176        'A <a href="http://reallylongdomainnamethatwillbetoolong.com/hi"'
177        ' title="http://reallylongdomainnamethatwillbetoolong.com/hi">http://reallylongdomainnametha...</a>!',  # noqa: E501
178    ),
179    (
180        "A file:///passwords.txt and http://web.com link",
181        {},
182        'A file:///passwords.txt and <a href="http://web.com">http://web.com</a> link',
183    ),
184    (
185        "A file:///passwords.txt and http://web.com link",
186        {"permitted_protocols": ["file"]},
187        'A <a href="file:///passwords.txt">file:///passwords.txt</a> and http://web.com link',
188    ),
189    (
190        "www.external-link.com",
191        {"extra_params": 'rel="nofollow" class="external"'},
192        '<a href="http://www.external-link.com" rel="nofollow" class="external">www.external-link.com</a>',  # noqa: E501
193    ),
194    (
195        "www.external-link.com and www.internal-link.com/blogs extra",
196        {
197            "extra_params": lambda href: (
198                'class="internal"'
199                if href.startswith("http://www.internal-link.com")
200                else 'rel="nofollow" class="external"'
201            )
202        },
203        '<a href="http://www.external-link.com" rel="nofollow" class="external">www.external-link.com</a>'  # noqa: E501
204        ' and <a href="http://www.internal-link.com/blogs" class="internal">www.internal-link.com/blogs</a> extra',  # noqa: E501
205    ),
206    (
207        "www.external-link.com",
208        {"extra_params": lambda href: '    rel="nofollow" class="external"  '},
209        '<a href="http://www.external-link.com" rel="nofollow" class="external">www.external-link.com</a>',  # noqa: E501
210    ),
211]  # type: List[Tuple[Union[str, bytes], Dict[str, Any], str]]
212
213
214class EscapeTestCase(unittest.TestCase):
215    def test_linkify(self):
216        for text, kwargs, html in linkify_tests:
217            linked = tornado.escape.linkify(text, **kwargs)
218            self.assertEqual(linked, html)
219
220    def test_xhtml_escape(self):
221        tests = [
222            ("<foo>", "&lt;foo&gt;"),
223            ("<foo>", "&lt;foo&gt;"),
224            (b"<foo>", b"&lt;foo&gt;"),
225            ("<>&\"'", "&lt;&gt;&amp;&quot;&#x27;"),
226            ("&amp;", "&amp;amp;"),
227            ("<\u00e9>", "&lt;\u00e9&gt;"),
228            (b"<\xc3\xa9>", b"&lt;\xc3\xa9&gt;"),
229        ]  # type: List[Tuple[Union[str, bytes], Union[str, bytes]]]
230        for unescaped, escaped in tests:
231            self.assertEqual(utf8(xhtml_escape(unescaped)), utf8(escaped))
232            self.assertEqual(utf8(unescaped), utf8(xhtml_unescape(escaped)))
233
234    def test_xhtml_unescape_numeric(self):
235        tests = [
236            ("foo&#32;bar", "foo bar"),
237            ("foo&#x20;bar", "foo bar"),
238            ("foo&#X20;bar", "foo bar"),
239            ("foo&#xabc;bar", "foo\u0abcbar"),
240            ("foo&#xyz;bar", "foo&#xyz;bar"),  # invalid encoding
241            ("foo&#;bar", "foo&#;bar"),  # invalid encoding
242            ("foo&#x;bar", "foo&#x;bar"),  # invalid encoding
243        ]
244        for escaped, unescaped in tests:
245            self.assertEqual(unescaped, xhtml_unescape(escaped))
246
247    def test_url_escape_unicode(self):
248        tests = [
249            # byte strings are passed through as-is
250            ("\u00e9".encode(), "%C3%A9"),
251            ("\u00e9".encode("latin1"), "%E9"),
252            # unicode strings become utf8
253            ("\u00e9", "%C3%A9"),
254        ]  # type: List[Tuple[Union[str, bytes], str]]
255        for unescaped, escaped in tests:
256            self.assertEqual(url_escape(unescaped), escaped)
257
258    def test_url_unescape_unicode(self):
259        tests = [
260            ("%C3%A9", "\u00e9", "utf8"),
261            ("%C3%A9", "\u00c3\u00a9", "latin1"),
262            ("%C3%A9", utf8("\u00e9"), None),
263        ]
264        for escaped, unescaped, encoding in tests:
265            # input strings to url_unescape should only contain ascii
266            # characters, but make sure the function accepts both byte
267            # and unicode strings.
268            self.assertEqual(url_unescape(to_unicode(escaped), encoding), unescaped)
269            self.assertEqual(url_unescape(utf8(escaped), encoding), unescaped)
270
271    def test_url_escape_quote_plus(self):
272        unescaped = "+ #%"
273        plus_escaped = "%2B+%23%25"
274        escaped = "%2B%20%23%25"
275        self.assertEqual(url_escape(unescaped), plus_escaped)
276        self.assertEqual(url_escape(unescaped, plus=False), escaped)
277        self.assertEqual(url_unescape(plus_escaped), unescaped)
278        self.assertEqual(url_unescape(escaped, plus=False), unescaped)
279        self.assertEqual(url_unescape(plus_escaped, encoding=None), utf8(unescaped))
280        self.assertEqual(
281            url_unescape(escaped, encoding=None, plus=False), utf8(unescaped)
282        )
283
284    def test_escape_return_types(self):
285        # On python2 the escape methods should generally return the same
286        # type as their argument
287        self.assertEqual(type(xhtml_escape("foo")), str)
288        self.assertEqual(type(xhtml_escape("foo")), unicode_type)
289
290    def test_json_decode(self):
291        # json_decode accepts both bytes and unicode, but strings it returns
292        # are always unicode.
293        self.assertEqual(json_decode(b'"foo"'), "foo")
294        self.assertEqual(json_decode('"foo"'), "foo")
295
296        # Non-ascii bytes are interpreted as utf8
297        self.assertEqual(json_decode(utf8('"\u00e9"')), "\u00e9")
298
299    def test_json_encode(self):
300        # json deals with strings, not bytes.  On python 2 byte strings will
301        # convert automatically if they are utf8; on python 3 byte strings
302        # are not allowed.
303        self.assertEqual(json_decode(json_encode("\u00e9")), "\u00e9")
304        if bytes is str:
305            self.assertEqual(json_decode(json_encode(utf8("\u00e9"))), "\u00e9")
306            self.assertRaises(UnicodeDecodeError, json_encode, b"\xe9")
307
308    def test_squeeze(self):
309        self.assertEqual(
310            squeeze("sequences     of    whitespace   chars"),
311            "sequences of whitespace chars",
312        )
313
314    def test_recursive_unicode(self):
315        tests = {
316            "dict": {b"foo": b"bar"},
317            "list": [b"foo", b"bar"],
318            "tuple": (b"foo", b"bar"),
319            "bytes": b"foo",
320        }
321        self.assertEqual(recursive_unicode(tests["dict"]), {"foo": "bar"})
322        self.assertEqual(recursive_unicode(tests["list"]), ["foo", "bar"])
323        self.assertEqual(recursive_unicode(tests["tuple"]), ("foo", "bar"))
324        self.assertEqual(recursive_unicode(tests["bytes"]), "foo")
325 
codekingpro/portable-devtools · Team Ai