codekingpro/portable-devtools
114k
1import unittest
2
3import tornado
4from tornado.escape import (
5 utf8,
6 xhtml_escape,
7 xhtml_unescape,
8 url_escape,
9 url_unescape,
10 to_unicode,
11 json_decode,
12 json_encode,
13 squeeze,
14 recursive_unicode,
15)
16from tornado.util import unicode_type
17
18from typing import List, Tuple, Union, Dict, Any # noqa: F401
19
20linkify_tests = [
21 # (input, linkify_kwargs, expected_output)
22 (
23 "hello http://world.com/!",
24 {},
25 'hello <a href="http://world.com/">http://world.com/</a>!',
26 ),
27 (
28 "hello http://world.com/with?param=true&stuff=yes",
29 {},
30 'hello <a href="http://world.com/with?param=true&stuff=yes">http://world.com/with?param=true&stuff=yes</a>', # noqa: E501
31 ),
32 # an opened paren followed by many chars killed Gruber's regex
33 (
34 "http://url.com/w(aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
35 {},
36 '<a href="http://url.com/w">http://url.com/w</a>(aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa', # noqa: E501
37 ),
38 # as did too many dots at the end
39 (
40 "http://url.com/withmany.......................................",
41 {},
42 '<a href="http://url.com/withmany">http://url.com/withmany</a>.......................................', # noqa: E501
43 ),
44 (
45 "http://url.com/withmany((((((((((((((((((((((((((((((((((a)",
46 {},
47 '<a href="http://url.com/withmany">http://url.com/withmany</a>((((((((((((((((((((((((((((((((((a)', # noqa: E501
48 ),
49 # some examples from http://daringfireball.net/2009/11/liberal_regex_for_matching_urls
50 # plus a fex extras (such as multiple parentheses).
51 (
52 "http://foo.com/blah_blah",
53 {},
54 '<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>',
55 ),
56 (
57 "http://foo.com/blah_blah/",
58 {},
59 '<a href="http://foo.com/blah_blah/">http://foo.com/blah_blah/</a>',
60 ),
61 (
62 "(Something like http://foo.com/blah_blah)",
63 {},
64 '(Something like <a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>)',
65 ),
66 (
67 "http://foo.com/blah_blah_(wikipedia)",
68 {},
69 '<a href="http://foo.com/blah_blah_(wikipedia)">http://foo.com/blah_blah_(wikipedia)</a>',
70 ),
71 (
72 "http://foo.com/blah_(blah)_(wikipedia)_blah",
73 {},
74 '<a href="http://foo.com/blah_(blah)_(wikipedia)_blah">http://foo.com/blah_(blah)_(wikipedia)_blah</a>', # noqa: E501
75 ),
76 (
77 "(Something like http://foo.com/blah_blah_(wikipedia))",
78 {},
79 '(Something like <a href="http://foo.com/blah_blah_(wikipedia)">http://foo.com/blah_blah_(wikipedia)</a>)', # noqa: E501
80 ),
81 (
82 "http://foo.com/blah_blah.",
83 {},
84 '<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>.',
85 ),
86 (
87 "http://foo.com/blah_blah/.",
88 {},
89 '<a href="http://foo.com/blah_blah/">http://foo.com/blah_blah/</a>.',
90 ),
91 (
92 "<http://foo.com/blah_blah>",
93 {},
94 '<<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>>',
95 ),
96 (
97 "<http://foo.com/blah_blah/>",
98 {},
99 '<<a href="http://foo.com/blah_blah/">http://foo.com/blah_blah/</a>>',
100 ),
101 (
102 "http://foo.com/blah_blah,",
103 {},
104 '<a href="http://foo.com/blah_blah">http://foo.com/blah_blah</a>,',
105 ),
106 (
107 "http://www.example.com/wpstyle/?p=364.",
108 {},
109 '<a href="http://www.example.com/wpstyle/?p=364">http://www.example.com/wpstyle/?p=364</a>.', # noqa: E501
110 ),
111 (
112 "rdar://1234",
113 {"permitted_protocols": ["http", "rdar"]},
114 '<a href="rdar://1234">rdar://1234</a>',
115 ),
116 (
117 "rdar:/1234",
118 {"permitted_protocols": ["rdar"]},
119 '<a href="rdar:/1234">rdar:/1234</a>',
120 ),
121 (
122 "http://userid:password@example.com:8080",
123 {},
124 '<a href="http://userid:password@example.com:8080">http://userid:password@example.com:8080</a>', # noqa: E501
125 ),
126 (
127 "http://userid@example.com",
128 {},
129 '<a href="http://userid@example.com">http://userid@example.com</a>',
130 ),
131 (
132 "http://userid@example.com:8080",
133 {},
134 '<a href="http://userid@example.com:8080">http://userid@example.com:8080</a>',
135 ),
136 (
137 "http://userid:password@example.com",
138 {},
139 '<a href="http://userid:password@example.com">http://userid:password@example.com</a>',
140 ),
141 (
142 "message://%3c330e7f8409726r6a4ba78dkf1fd71420c1bf6ff@mail.gmail.com%3e",
143 {"permitted_protocols": ["http", "message"]},
144 '<a href="message://%3c330e7f8409726r6a4ba78dkf1fd71420c1bf6ff@mail.gmail.com%3e">'
145 "message://%3c330e7f8409726r6a4ba78dkf1fd71420c1bf6ff@mail.gmail.com%3e</a>",
146 ),
147 (
148 "http://\u27a1.ws/\u4a39",
149 {},
150 '<a href="http://\u27a1.ws/\u4a39">http://\u27a1.ws/\u4a39</a>',
151 ),
152 (
153 "<tag>http://example.com</tag>",
154 {},
155 '<tag><a href="http://example.com">http://example.com</a></tag>',
156 ),
157 (
158 "Just a www.example.com link.",
159 {},
160 'Just a <a href="http://www.example.com">www.example.com</a> link.',
161 ),
162 (
163 "Just a www.example.com link.",
164 {"require_protocol": True},
165 "Just a www.example.com link.",
166 ),
167 (
168 "A http://reallylong.com/link/that/exceedsthelenglimit.html",
169 {"require_protocol": True, "shorten": True},
170 'A <a href="http://reallylong.com/link/that/exceedsthelenglimit.html"'
171 ' title="http://reallylong.com/link/that/exceedsthelenglimit.html">http://reallylong.com/link...</a>', # noqa: E501
172 ),
173 (
174 "A http://reallylongdomainnamethatwillbetoolong.com/hi!",
175 {"shorten": True},
176 'A <a href="http://reallylongdomainnamethatwillbetoolong.com/hi"'
177 ' title="http://reallylongdomainnamethatwillbetoolong.com/hi">http://reallylongdomainnametha...</a>!', # noqa: E501
178 ),
179 (
180 "A file:///passwords.txt and http://web.com link",
181 {},
182 'A file:///passwords.txt and <a href="http://web.com">http://web.com</a> link',
183 ),
184 (
185 "A file:///passwords.txt and http://web.com link",
186 {"permitted_protocols": ["file"]},
187 'A <a href="file:///passwords.txt">file:///passwords.txt</a> and http://web.com link',
188 ),
189 (
190 "www.external-link.com",
191 {"extra_params": 'rel="nofollow" class="external"'},
192 '<a href="http://www.external-link.com" rel="nofollow" class="external">www.external-link.com</a>', # noqa: E501
193 ),
194 (
195 "www.external-link.com and www.internal-link.com/blogs extra",
196 {
197 "extra_params": lambda href: (
198 'class="internal"'
199 if href.startswith("http://www.internal-link.com")
200 else 'rel="nofollow" class="external"'
201 )
202 },
203 '<a href="http://www.external-link.com" rel="nofollow" class="external">www.external-link.com</a>' # noqa: E501
204 ' and <a href="http://www.internal-link.com/blogs" class="internal">www.internal-link.com/blogs</a> extra', # noqa: E501
205 ),
206 (
207 "www.external-link.com",
208 {"extra_params": lambda href: ' rel="nofollow" class="external" '},
209 '<a href="http://www.external-link.com" rel="nofollow" class="external">www.external-link.com</a>', # noqa: E501
210 ),
211] # type: List[Tuple[Union[str, bytes], Dict[str, Any], str]]
212
213
214class EscapeTestCase(unittest.TestCase):
215 def test_linkify(self):
216 for text, kwargs, html in linkify_tests:
217 linked = tornado.escape.linkify(text, **kwargs)
218 self.assertEqual(linked, html)
219
220 def test_xhtml_escape(self):
221 tests = [
222 ("<foo>", "<foo>"),
223 ("<foo>", "<foo>"),
224 (b"<foo>", b"<foo>"),
225 ("<>&\"'", "<>&"'"),
226 ("&", "&amp;"),
227 ("<\u00e9>", "<\u00e9>"),
228 (b"<\xc3\xa9>", b"<\xc3\xa9>"),
229 ] # type: List[Tuple[Union[str, bytes], Union[str, bytes]]]
230 for unescaped, escaped in tests:
231 self.assertEqual(utf8(xhtml_escape(unescaped)), utf8(escaped))
232 self.assertEqual(utf8(unescaped), utf8(xhtml_unescape(escaped)))
233
234 def test_xhtml_unescape_numeric(self):
235 tests = [
236 ("foo bar", "foo bar"),
237 ("foo bar", "foo bar"),
238 ("foo bar", "foo bar"),
239 ("foo઼bar", "foo\u0abcbar"),
240 ("foo&#xyz;bar", "foo&#xyz;bar"), # invalid encoding
241 ("foo&#;bar", "foo&#;bar"), # invalid encoding
242 ("foo&#x;bar", "foo&#x;bar"), # invalid encoding
243 ]
244 for escaped, unescaped in tests:
245 self.assertEqual(unescaped, xhtml_unescape(escaped))
246
247 def test_url_escape_unicode(self):
248 tests = [
249 # byte strings are passed through as-is
250 ("\u00e9".encode(), "%C3%A9"),
251 ("\u00e9".encode("latin1"), "%E9"),
252 # unicode strings become utf8
253 ("\u00e9", "%C3%A9"),
254 ] # type: List[Tuple[Union[str, bytes], str]]
255 for unescaped, escaped in tests:
256 self.assertEqual(url_escape(unescaped), escaped)
257
258 def test_url_unescape_unicode(self):
259 tests = [
260 ("%C3%A9", "\u00e9", "utf8"),
261 ("%C3%A9", "\u00c3\u00a9", "latin1"),
262 ("%C3%A9", utf8("\u00e9"), None),
263 ]
264 for escaped, unescaped, encoding in tests:
265 # input strings to url_unescape should only contain ascii
266 # characters, but make sure the function accepts both byte
267 # and unicode strings.
268 self.assertEqual(url_unescape(to_unicode(escaped), encoding), unescaped)
269 self.assertEqual(url_unescape(utf8(escaped), encoding), unescaped)
270
271 def test_url_escape_quote_plus(self):
272 unescaped = "+ #%"
273 plus_escaped = "%2B+%23%25"
274 escaped = "%2B%20%23%25"
275 self.assertEqual(url_escape(unescaped), plus_escaped)
276 self.assertEqual(url_escape(unescaped, plus=False), escaped)
277 self.assertEqual(url_unescape(plus_escaped), unescaped)
278 self.assertEqual(url_unescape(escaped, plus=False), unescaped)
279 self.assertEqual(url_unescape(plus_escaped, encoding=None), utf8(unescaped))
280 self.assertEqual(
281 url_unescape(escaped, encoding=None, plus=False), utf8(unescaped)
282 )
283
284 def test_escape_return_types(self):
285 # On python2 the escape methods should generally return the same
286 # type as their argument
287 self.assertEqual(type(xhtml_escape("foo")), str)
288 self.assertEqual(type(xhtml_escape("foo")), unicode_type)
289
290 def test_json_decode(self):
291 # json_decode accepts both bytes and unicode, but strings it returns
292 # are always unicode.
293 self.assertEqual(json_decode(b'"foo"'), "foo")
294 self.assertEqual(json_decode('"foo"'), "foo")
295
296 # Non-ascii bytes are interpreted as utf8
297 self.assertEqual(json_decode(utf8('"\u00e9"')), "\u00e9")
298
299 def test_json_encode(self):
300 # json deals with strings, not bytes. On python 2 byte strings will
301 # convert automatically if they are utf8; on python 3 byte strings
302 # are not allowed.
303 self.assertEqual(json_decode(json_encode("\u00e9")), "\u00e9")
304 if bytes is str:
305 self.assertEqual(json_decode(json_encode(utf8("\u00e9"))), "\u00e9")
306 self.assertRaises(UnicodeDecodeError, json_encode, b"\xe9")
307
308 def test_squeeze(self):
309 self.assertEqual(
310 squeeze("sequences of whitespace chars"),
311 "sequences of whitespace chars",
312 )
313
314 def test_recursive_unicode(self):
315 tests = {
316 "dict": {b"foo": b"bar"},
317 "list": [b"foo", b"bar"],
318 "tuple": (b"foo", b"bar"),
319 "bytes": b"foo",
320 }
321 self.assertEqual(recursive_unicode(tests["dict"]), {"foo": "bar"})
322 self.assertEqual(recursive_unicode(tests["list"]), ["foo", "bar"])
323 self.assertEqual(recursive_unicode(tests["tuple"]), ("foo", "bar"))
324 self.assertEqual(recursive_unicode(tests["bytes"]), "foo")
325 