codekingpro/portable-devtools
115k
1import operator
2import sys
3
4import pytest
5
6import numpy as np
7from numpy._core._exceptions import _UFuncNoLoopError
8from numpy.testing import IS_PYPY, assert_array_equal, assert_raises
9from numpy.testing._private.utils import requires_memory
10
11COMPARISONS = [
12 (operator.eq, np.equal, "=="),
13 (operator.ne, np.not_equal, "!="),
14 (operator.lt, np.less, "<"),
15 (operator.le, np.less_equal, "<="),
16 (operator.gt, np.greater, ">"),
17 (operator.ge, np.greater_equal, ">="),
18]
19
20MAX = np.iinfo(np.int64).max
21
22IS_PYPY_LT_7_3_16 = IS_PYPY and sys.implementation.version < (7, 3, 16)
23
24@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
25def test_mixed_string_comparison_ufuncs_fail(op, ufunc, sym):
26 arr_string = np.array(["a", "b"], dtype="S")
27 arr_unicode = np.array(["a", "c"], dtype="U")
28
29 with pytest.raises(TypeError, match="did not contain a loop"):
30 ufunc(arr_string, arr_unicode)
31
32 with pytest.raises(TypeError, match="did not contain a loop"):
33 ufunc(arr_unicode, arr_string)
34
35@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
36def test_mixed_string_comparisons_ufuncs_with_cast(op, ufunc, sym):
37 arr_string = np.array(["a", "b"], dtype="S")
38 arr_unicode = np.array(["a", "c"], dtype="U")
39
40 # While there is no loop, manual casting is acceptable:
41 res1 = ufunc(arr_string, arr_unicode, signature="UU->?", casting="unsafe")
42 res2 = ufunc(arr_string, arr_unicode, signature="SS->?", casting="unsafe")
43
44 expected = op(arr_string.astype("U"), arr_unicode)
45 assert_array_equal(res1, expected)
46 assert_array_equal(res2, expected)
47
48
49@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
50@pytest.mark.parametrize("dtypes", [
51 ("S2", "S2"), ("S2", "S10"),
52 ("<U1", "<U1"), ("<U1", ">U1"), (">U1", ">U1"),
53 ("<U1", "<U10"), ("<U1", ">U10")])
54@pytest.mark.parametrize("aligned", [True, False])
55def test_string_comparisons(op, ufunc, sym, dtypes, aligned):
56 # ensure native byte-order for the first view to stay within unicode range
57 native_dt = np.dtype(dtypes[0]).newbyteorder("=")
58 arr = np.arange(2**15).view(native_dt).astype(dtypes[0])
59 if not aligned:
60 # Make `arr` unaligned:
61 new = np.zeros(arr.nbytes + 1, dtype=np.uint8)[1:].view(dtypes[0])
62 new[...] = arr
63 arr = new
64
65 arr2 = arr.astype(dtypes[1], copy=True)
66 np.random.shuffle(arr2)
67 arr[0] = arr2[0] # make sure one matches
68
69 expected = [op(d1, d2) for d1, d2 in zip(arr.tolist(), arr2.tolist())]
70 assert_array_equal(op(arr, arr2), expected)
71 assert_array_equal(ufunc(arr, arr2), expected)
72 assert_array_equal(
73 np.char.compare_chararrays(arr, arr2, sym, False), expected
74 )
75
76 expected = [op(d2, d1) for d1, d2 in zip(arr.tolist(), arr2.tolist())]
77 assert_array_equal(op(arr2, arr), expected)
78 assert_array_equal(ufunc(arr2, arr), expected)
79 assert_array_equal(
80 np.char.compare_chararrays(arr2, arr, sym, False), expected
81 )
82
83
84@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
85@pytest.mark.parametrize("dtypes", [
86 ("S2", "S2"), ("S2", "S10"), ("<U1", "<U1"), ("<U1", ">U10")])
87def test_string_comparisons_empty(op, ufunc, sym, dtypes):
88 arr = np.empty((1, 0, 1, 5), dtype=dtypes[0])
89 arr2 = np.empty((100, 1, 0, 1), dtype=dtypes[1])
90
91 expected = np.empty(np.broadcast_shapes(arr.shape, arr2.shape), dtype=bool)
92 assert_array_equal(op(arr, arr2), expected)
93 assert_array_equal(ufunc(arr, arr2), expected)
94 assert_array_equal(
95 np.char.compare_chararrays(arr, arr2, sym, False), expected
96 )
97
98
99@pytest.mark.parametrize("str_dt", ["S", "U"])
100@pytest.mark.parametrize("float_dt", np.typecodes["AllFloat"])
101def test_float_to_string_cast(str_dt, float_dt):
102 float_dt = np.dtype(float_dt)
103 fi = np.finfo(float_dt)
104 arr = np.array([np.nan, np.inf, -np.inf, fi.max, fi.min], dtype=float_dt)
105 expected = ["nan", "inf", "-inf", str(fi.max), str(fi.min)]
106 if float_dt.kind == "c":
107 expected = [f"({r}+0j)" for r in expected]
108
109 res = arr.astype(str_dt)
110 assert_array_equal(res, np.array(expected, dtype=str_dt))
111
112
113@pytest.mark.parametrize("str_dt", "US")
114@pytest.mark.parametrize("size", [-1, np.iinfo(np.intc).max])
115def test_string_size_dtype_errors(str_dt, size):
116 if size > 0:
117 size = size // np.dtype(f"{str_dt}1").itemsize + 1
118
119 with pytest.raises(ValueError):
120 np.dtype((str_dt, size))
121 with pytest.raises(TypeError):
122 np.dtype(f"{str_dt}{size}")
123
124
125@pytest.mark.parametrize("str_dt", "US")
126def test_string_size_dtype_large_repr(str_dt):
127 size = np.iinfo(np.intc).max // np.dtype(f"{str_dt}1").itemsize
128 size_str = str(size)
129
130 dtype = np.dtype((str_dt, size))
131 assert size_str in dtype.str
132 assert size_str in str(dtype)
133 assert size_str in repr(dtype)
134
135
136@pytest.mark.slow
137@requires_memory(2 * np.iinfo(np.intc).max)
138@pytest.mark.parametrize("str_dt", "US")
139@pytest.mark.thread_unsafe(reason="crashes with low memory")
140def test_large_string_coercion_error(str_dt):
141 very_large = np.iinfo(np.intc).max // np.dtype(f"{str_dt}1").itemsize
142 try:
143 large_string = "A" * (very_large + 1)
144 except Exception:
145 # We may not be able to create this Python string on 32bit.
146 pytest.skip("python failed to create huge string")
147
148 class MyStr:
149 def __str__(self):
150 return large_string
151
152 try:
153 # TypeError from NumPy, or OverflowError from 32bit Python.
154 with pytest.raises((TypeError, OverflowError)):
155 np.array([large_string], dtype=str_dt)
156
157 # Same as above, but input has to be converted to a string.
158 with pytest.raises((TypeError, OverflowError)):
159 np.array([MyStr()], dtype=str_dt)
160 except MemoryError:
161 # Catch memory errors, because `requires_memory` would do so.
162 raise AssertionError("Ops should raise before any large allocation.")
163
164@pytest.mark.slow
165@requires_memory(2 * np.iinfo(np.intc).max)
166@pytest.mark.parametrize("str_dt", "US")
167@pytest.mark.thread_unsafe(reason="crashes with low memory")
168def test_large_string_addition_error(str_dt):
169 very_large = np.iinfo(np.intc).max // np.dtype(f"{str_dt}1").itemsize
170
171 a = np.array(["A" * very_large], dtype=str_dt)
172 b = np.array("B", dtype=str_dt)
173 try:
174 with pytest.raises(TypeError):
175 np.add(a, b)
176 with pytest.raises(TypeError):
177 np.add(a, a)
178 except MemoryError:
179 # Catch memory errors, because `requires_memory` would do so.
180 raise AssertionError("Ops should raise before any large allocation.")
181
182
183def test_large_string_cast():
184 very_large = np.iinfo(np.intc).max // 4
185 # Could be nice to test very large path, but it makes too many huge
186 # allocations right now (need non-legacy cast loops for this).
187 # a = np.array([], dtype=np.dtype(("S", very_large)))
188 # assert a.astype("U").dtype.itemsize == very_large * 4
189
190 a = np.array([], dtype=np.dtype(("S", very_large + 1)))
191 # It is not perfect but OK if this raises a MemoryError during setup
192 # (this happens due clunky code and/or buffer setup.)
193 with pytest.raises((TypeError, MemoryError)):
194 a.astype("U")
195
196
197@pytest.mark.parametrize("dt", ["S1", "U1"])
198def test_in_place_mutiply_no_overflow(dt):
199 # see gh-30495
200 a = np.array("a", dtype=dt)
201 a *= 20
202 assert_array_equal(a, np.array("a", dtype=dt))
203
204
205@pytest.mark.parametrize("dt", ["S", "U", "T"])
206class TestMethods:
207
208 @pytest.mark.parametrize("in1,in2,out", [
209 ("", "", ""),
210 ("abc", "abc", "abcabc"),
211 ("12345", "12345", "1234512345"),
212 ("MixedCase", "MixedCase", "MixedCaseMixedCase"),
213 ("12345 \0 ", "12345 \0 ", "12345 \0 12345 \0 "),
214 ("UPPER", "UPPER", "UPPERUPPER"),
215 (["abc", "def"], ["hello", "world"], ["abchello", "defworld"]),
216 ])
217 def test_add(self, in1, in2, out, dt):
218 in1 = np.array(in1, dtype=dt)
219 in2 = np.array(in2, dtype=dt)
220 out = np.array(out, dtype=dt)
221 assert_array_equal(np.strings.add(in1, in2), out)
222
223 @pytest.mark.parametrize("in1,in2,out", [
224 ("abc", 3, "abcabcabc"),
225 ("abc", 0, ""),
226 ("abc", -1, ""),
227 (["abc", "def"], [1, 4], ["abc", "defdefdefdef"]),
228 ])
229 def test_multiply(self, in1, in2, out, dt):
230 in1 = np.array(in1, dtype=dt)
231 out = np.array(out, dtype=dt)
232 assert_array_equal(np.strings.multiply(in1, in2), out)
233
234 def test_multiply_raises(self, dt):
235 with pytest.raises(TypeError, match="unsupported type"):
236 np.strings.multiply(np.array("abc", dtype=dt), 3.14)
237
238 with pytest.raises(OverflowError):
239 np.strings.multiply(np.array("abc", dtype=dt), sys.maxsize)
240
241 def test_inplace_multiply(self, dt):
242 arr = np.array(['foo ', 'bar'], dtype=dt)
243 arr *= 2
244 if dt != "T":
245 assert_array_equal(arr, np.array(['foo ', 'barb'], dtype=dt))
246 else:
247 assert_array_equal(arr, ['foo foo ', 'barbar'])
248
249 with pytest.raises(OverflowError):
250 arr *= sys.maxsize
251
252 @pytest.mark.parametrize("i_dt", [np.int8, np.int16, np.int32,
253 np.int64, np.int_])
254 def test_multiply_integer_dtypes(self, i_dt, dt):
255 a = np.array("abc", dtype=dt)
256 i = np.array(3, dtype=i_dt)
257 res = np.array("abcabcabc", dtype=dt)
258 assert_array_equal(np.strings.multiply(a, i), res)
259
260 @pytest.mark.parametrize("in_,out", [
261 ("", False),
262 ("a", True),
263 ("A", True),
264 ("\n", False),
265 ("abc", True),
266 ("aBc123", False),
267 ("abc\n", False),
268 (["abc", "aBc123"], [True, False]),
269 ])
270 def test_isalpha(self, in_, out, dt):
271 in_ = np.array(in_, dtype=dt)
272 assert_array_equal(np.strings.isalpha(in_), out)
273
274 @pytest.mark.parametrize("in_,out", [
275 ('', False),
276 ('a', True),
277 ('A', True),
278 ('\n', False),
279 ('123abc456', True),
280 ('a1b3c', True),
281 ('aBc000 ', False),
282 ('abc\n', False),
283 ])
284 def test_isalnum(self, in_, out, dt):
285 in_ = np.array(in_, dtype=dt)
286 assert_array_equal(np.strings.isalnum(in_), out)
287
288 @pytest.mark.parametrize("in_,out", [
289 ("", False),
290 ("a", False),
291 ("0", True),
292 ("012345", True),
293 ("012345a", False),
294 (["a", "012345"], [False, True]),
295 ])
296 def test_isdigit(self, in_, out, dt):
297 in_ = np.array(in_, dtype=dt)
298 assert_array_equal(np.strings.isdigit(in_), out)
299
300 @pytest.mark.parametrize("in_,out", [
301 ("", False),
302 ("a", False),
303 ("1", False),
304 (" ", True),
305 ("\t", True),
306 ("\r", True),
307 ("\n", True),
308 (" \t\r \n", True),
309 (" \t\r\na", False),
310 (["\t1", " \t\r \n"], [False, True])
311 ])
312 def test_isspace(self, in_, out, dt):
313 in_ = np.array(in_, dtype=dt)
314 assert_array_equal(np.strings.isspace(in_), out)
315
316 @pytest.mark.parametrize("in_,out", [
317 ('', False),
318 ('a', True),
319 ('A', False),
320 ('\n', False),
321 ('abc', True),
322 ('aBc', False),
323 ('abc\n', True),
324 ])
325 def test_islower(self, in_, out, dt):
326 in_ = np.array(in_, dtype=dt)
327 assert_array_equal(np.strings.islower(in_), out)
328
329 @pytest.mark.parametrize("in_,out", [
330 ('', False),
331 ('a', False),
332 ('A', True),
333 ('\n', False),
334 ('ABC', True),
335 ('AbC', False),
336 ('ABC\n', True),
337 ])
338 def test_isupper(self, in_, out, dt):
339 in_ = np.array(in_, dtype=dt)
340 assert_array_equal(np.strings.isupper(in_), out)
341
342 @pytest.mark.parametrize("in_,out", [
343 ('', False),
344 ('a', False),
345 ('A', True),
346 ('\n', False),
347 ('A Titlecased Line', True),
348 ('A\nTitlecased Line', True),
349 ('A Titlecased, Line', True),
350 ('Not a capitalized String', False),
351 ('Not\ta Titlecase String', False),
352 ('Not--a Titlecase String', False),
353 ('NOT', False),
354 ])
355 def test_istitle(self, in_, out, dt):
356 in_ = np.array(in_, dtype=dt)
357 assert_array_equal(np.strings.istitle(in_), out)
358
359 @pytest.mark.parametrize("in_,out", [
360 ("", 0),
361 ("abc", 3),
362 ("12345", 5),
363 ("MixedCase", 9),
364 ("12345 \x00 ", 8),
365 ("UPPER", 5),
366 (["abc", "12345 \x00 "], [3, 8]),
367 ])
368 def test_str_len(self, in_, out, dt):
369 in_ = np.array(in_, dtype=dt)
370 assert_array_equal(np.strings.str_len(in_), out)
371
372 @pytest.mark.parametrize("a,sub,start,end,out", [
373 ("abcdefghiabc", "abc", 0, None, 0),
374 ("abcdefghiabc", "abc", 1, None, 9),
375 ("abcdefghiabc", "def", 4, None, -1),
376 ("abc", "", 0, None, 0),
377 ("abc", "", 3, None, 3),
378 ("abc", "", 4, None, -1),
379 ("rrarrrrrrrrra", "a", 0, None, 2),
380 ("rrarrrrrrrrra", "a", 4, None, 12),
381 ("rrarrrrrrrrra", "a", 4, 6, -1),
382 ("", "", 0, None, 0),
383 ("", "", 1, 1, -1),
384 ("", "", MAX, 0, -1),
385 ("", "xx", 0, None, -1),
386 ("", "xx", 1, 1, -1),
387 ("", "xx", MAX, 0, -1),
388 pytest.param(99 * "a" + "b", "b", 0, None, 99,
389 id="99*a+b-b-0-None-99"),
390 pytest.param(98 * "a" + "ba", "ba", 0, None, 98,
391 id="98*a+ba-ba-0-None-98"),
392 pytest.param(100 * "a", "b", 0, None, -1,
393 id="100*a-b-0-None--1"),
394 pytest.param(30000 * "a" + 100 * "b", 100 * "b", 0, None, 30000,
395 id="30000*a+100*b-100*b-0-None-30000"),
396 pytest.param(30000 * "a", 100 * "b", 0, None, -1,
397 id="30000*a-100*b-0-None--1"),
398 pytest.param(15000 * "a" + 15000 * "b", 15000 * "b", 0, None, 15000,
399 id="15000*a+15000*b-15000*b-0-None-15000"),
400 pytest.param(15000 * "a" + 15000 * "b", 15000 * "c", 0, None, -1,
401 id="15000*a+15000*b-15000*c-0-None--1"),
402 (["abcdefghiabc", "rrarrrrrrrrra"], ["def", "arr"], [0, 3],
403 None, [3, -1]),
404 ("Ae¢☃€ 😊" * 2, "😊", 0, None, 6),
405 ("Ae¢☃€ 😊" * 2, "😊", 7, None, 13),
406 pytest.param("A" * (2 ** 17), r"[\w]+\Z", 0, None, -1,
407 id=r"A*2**17-[\w]+\Z-0-None--1"),
408 ])
409 def test_find(self, a, sub, start, end, out, dt):
410 if "😊" in a and dt == "S":
411 pytest.skip("Bytes dtype does not support non-ascii input")
412 a = np.array(a, dtype=dt)
413 sub = np.array(sub, dtype=dt)
414 assert_array_equal(np.strings.find(a, sub, start, end), out)
415
416 @pytest.mark.parametrize("a,sub,start,end,out", [
417 ("abcdefghiabc", "abc", 0, None, 9),
418 ("abcdefghiabc", "", 0, None, 12),
419 ("abcdefghiabc", "abcd", 0, None, 0),
420 ("abcdefghiabc", "abcz", 0, None, -1),
421 ("abc", "", 0, None, 3),
422 ("abc", "", 3, None, 3),
423 ("abc", "", 4, None, -1),
424 ("rrarrrrrrrrra", "a", 0, None, 12),
425 ("rrarrrrrrrrra", "a", 4, None, 12),
426 ("rrarrrrrrrrra", "a", 4, 6, -1),
427 (["abcdefghiabc", "rrarrrrrrrrra"], ["abc", "a"], [0, 0],
428 None, [9, 12]),
429 ("Ae¢☃€ 😊" * 2, "😊", 0, None, 13),
430 ("Ae¢☃€ 😊" * 2, "😊", 0, 7, 6),
431 ])
432 def test_rfind(self, a, sub, start, end, out, dt):
433 if "😊" in a and dt == "S":
434 pytest.skip("Bytes dtype does not support non-ascii input")
435 a = np.array(a, dtype=dt)
436 sub = np.array(sub, dtype=dt)
437 assert_array_equal(np.strings.rfind(a, sub, start, end), out)
438
439 @pytest.mark.parametrize("a,sub,start,end,out", [
440 ("aaa", "a", 0, None, 3),
441 ("aaa", "b", 0, None, 0),
442 ("aaa", "a", 1, None, 2),
443 ("aaa", "a", 10, None, 0),
444 ("aaa", "a", -1, None, 1),
445 ("aaa", "a", -10, None, 3),
446 ("aaa", "a", 0, 1, 1),
447 ("aaa", "a", 0, 10, 3),
448 ("aaa", "a", 0, -1, 2),
449 ("aaa", "a", 0, -10, 0),
450 ("aaa", "", 1, None, 3),
451 ("aaa", "", 3, None, 1),
452 ("aaa", "", 10, None, 0),
453 ("aaa", "", -1, None, 2),
454 ("aaa", "", -10, None, 4),
455 ("aaa", "aaaa", 0, None, 0),
456 pytest.param(98 * "a" + "ba", "ba", 0, None, 1,
457 id="98*a+ba-ba-0-None-1"),
458 pytest.param(30000 * "a" + 100 * "b", 100 * "b", 0, None, 1,
459 id="30000*a+100*b-100*b-0-None-1"),
460 pytest.param(30000 * "a", 100 * "b", 0, None, 0,
461 id="30000*a-100*b-0-None-0"),
462 pytest.param(30000 * "a" + 100 * "ab", "ab", 0, None, 100,
463 id="30000*a+100*ab-ab-0-None-100"),
464 pytest.param(15000 * "a" + 15000 * "b", 15000 * "b", 0, None, 1,
465 id="15000*a+15000*b-15000*b-0-None-1"),
466 pytest.param(15000 * "a" + 15000 * "b", 15000 * "c", 0, None, 0,
467 id="15000*a+15000*b-15000*c-0-None-0"),
468 ("", "", 0, None, 1),
469 ("", "", 1, 1, 0),
470 ("", "", MAX, 0, 0),
471 ("", "xx", 0, None, 0),
472 ("", "xx", 1, 1, 0),
473 ("", "xx", MAX, 0, 0),
474 (["aaa", ""], ["a", ""], [0, 0], None, [3, 1]),
475 ("Ae¢☃€ 😊" * 100, "😊", 0, None, 100),
476 ])
477 def test_count(self, a, sub, start, end, out, dt):
478 if "😊" in a and dt == "S":
479 pytest.skip("Bytes dtype does not support non-ascii input")
480 a = np.array(a, dtype=dt)
481 sub = np.array(sub, dtype=dt)
482 assert_array_equal(np.strings.count(a, sub, start, end), out)
483
484 @pytest.mark.parametrize("a,prefix,start,end,out", [
485 ("hello", "he", 0, None, True),
486 ("hello", "hello", 0, None, True),
487 ("hello", "hello world", 0, None, False),
488 ("hello", "", 0, None, True),
489 ("hello", "ello", 0, None, False),
490 ("hello", "ello", 1, None, True),
491 ("hello", "o", 4, None, True),
492 ("hello", "o", 5, None, False),
493 ("hello", "", 5, None, True),
494 ("hello", "lo", 6, None, False),
495 ("helloworld", "lowo", 3, None, True),
496 ("helloworld", "lowo", 3, 7, True),
497 ("helloworld", "lowo", 3, 6, False),
498 ("", "", 0, 1, True),
499 ("", "", 0, 0, True),
500 ("", "", 1, 0, False),
501 ("hello", "he", 0, -1, True),
502 ("hello", "he", -53, -1, True),
503 ("hello", "hello", 0, -1, False),
504 ("hello", "hello world", -1, -10, False),
505 ("hello", "ello", -5, None, False),
506 ("hello", "ello", -4, None, True),
507 ("hello", "o", -2, None, False),
508 ("hello", "o", -1, None, True),
509 ("hello", "", -3, -3, True),
510 ("hello", "lo", -9, None, False),
511 (["hello", ""], ["he", ""], [0, 0], None, [True, True]),
512 ])
513 def test_startswith(self, a, prefix, start, end, out, dt):
514 a = np.array(a, dtype=dt)
515 prefix = np.array(prefix, dtype=dt)
516 assert_array_equal(np.strings.startswith(a, prefix, start, end), out)
517
518 @pytest.mark.parametrize("a,suffix,start,end,out", [
519 ("hello", "lo", 0, None, True),
520 ("hello", "he", 0, None, False),
521 ("hello", "", 0, None, True),
522 ("hello", "hello world", 0, None, False),
523 ("helloworld", "worl", 0, None, False),
524 ("helloworld", "worl", 3, 9, True),
525 ("helloworld", "world", 3, 12, True),
526 ("helloworld", "lowo", 1, 7, True),
527 ("helloworld", "lowo", 2, 7, True),
528 ("helloworld", "lowo", 3, 7, True),
529 ("helloworld", "lowo", 4, 7, False),
530 ("helloworld", "lowo", 3, 8, False),
531 ("ab", "ab", 0, 1, False),
532 ("ab", "ab", 0, 0, False),
533 ("", "", 0, 1, True),
534 ("", "", 0, 0, True),
535 ("", "", 1, 0, False),
536 ("hello", "lo", -2, None, True),
537 ("hello", "he", -2, None, False),
538 ("hello", "", -3, -3, True),
539 ("hello", "hello world", -10, -2, False),
540 ("helloworld", "worl", -6, None, False),
541 ("helloworld", "worl", -5, -1, True),
542 ("helloworld", "worl", -5, 9, True),
543 ("helloworld", "world", -7, 12, True),
544 ("helloworld", "lowo", -99, -3, True),
545 ("helloworld", "lowo", -8, -3, True),
546 ("helloworld", "lowo", -7, -3, True),
547 ("helloworld", "lowo", 3, -4, False),
548 ("helloworld", "lowo", -8, -2, False),
549 (["hello", "helloworld"], ["lo", "worl"], [0, -6], None,
550 [True, False]),
551 ])
552 def test_endswith(self, a, suffix, start, end, out, dt):
553 a = np.array(a, dtype=dt)
554 suffix = np.array(suffix, dtype=dt)
555 assert_array_equal(np.strings.endswith(a, suffix, start, end), out)
556
557 @pytest.mark.parametrize("a,chars,out", [
558 ("", None, ""),
559 (" hello ", None, "hello "),
560 ("hello", None, "hello"),
561 (" \t\n\r\f\vabc \t\n\r\f\v", None, "abc \t\n\r\f\v"),
562 ([" hello ", "hello"], None, ["hello ", "hello"]),
563 ("", "", ""),
564 ("", "xyz", ""),
565 ("hello", "", "hello"),
566 ("xyzzyhelloxyzzy", "xyz", "helloxyzzy"),
567 ("hello", "xyz", "hello"),
568 ("xyxz", "xyxz", ""),
569 ("xyxzx", "x", "yxzx"),
570 (["xyzzyhelloxyzzy", "hello"], ["xyz", "xyz"],
571 ["helloxyzzy", "hello"]),
572 (["ba", "ac", "baa", "bba"], "b", ["a", "ac", "aa", "a"]),
573 ])
574 def test_lstrip(self, a, chars, out, dt):
575 a = np.array(a, dtype=dt)
576 out = np.array(out, dtype=dt)
577 if chars is not None:
578 chars = np.array(chars, dtype=dt)
579 assert_array_equal(np.strings.lstrip(a, chars), out)
580 else:
581 assert_array_equal(np.strings.lstrip(a), out)
582
583 @pytest.mark.parametrize("a,chars,out", [
584 ("", None, ""),
585 (" hello ", None, " hello"),
586 ("hello", None, "hello"),
587 (" \t\n\r\f\vabc \t\n\r\f\v", None, " \t\n\r\f\vabc"),
588 ([" hello ", "hello"], None, [" hello", "hello"]),
589 ("", "", ""),
590 ("", "xyz", ""),
591 ("hello", "", "hello"),
592 (["hello ", "abcdefghijklmnop"], None,
593 ["hello", "abcdefghijklmnop"]),
594 ("xyzzyhelloxyzzy", "xyz", "xyzzyhello"),
595 ("hello", "xyz", "hello"),
596 ("xyxz", "xyxz", ""),
597 (" ", None, ""),
598 ("xyxzx", "x", "xyxz"),
599 (["xyzzyhelloxyzzy", "hello"], ["xyz", "xyz"],
600 ["xyzzyhello", "hello"]),
601 (["ab", "ac", "aab", "abb"], "b", ["a", "ac", "aa", "a"]),
602 ])
603 def test_rstrip(self, a, chars, out, dt):
604 a = np.array(a, dtype=dt)
605 out = np.array(out, dtype=dt)
606 if chars is not None:
607 chars = np.array(chars, dtype=dt)
608 assert_array_equal(np.strings.rstrip(a, chars), out)
609 else:
610 assert_array_equal(np.strings.rstrip(a), out)
611
612 @pytest.mark.parametrize("a,chars,out", [
613 ("", None, ""),
614 (" hello ", None, "hello"),
615 ("hello", None, "hello"),
616 (" \t\n\r\f\vabc \t\n\r\f\v", None, "abc"),
617 ([" hello ", "hello"], None, ["hello", "hello"]),
618 ("", "", ""),
619 ("", "xyz", ""),
620 ("hello", "", "hello"),
621 ("xyzzyhelloxyzzy", "xyz", "hello"),
622 ("hello", "xyz", "hello"),
623 ("xyxz", "xyxz", ""),
624 ("xyxzx", "x", "yxz"),
625 (["xyzzyhelloxyzzy", "hello"], ["xyz", "xyz"],
626 ["hello", "hello"]),
627 (["bab", "ac", "baab", "bbabb"], "b", ["a", "ac", "aa", "a"]),
628 ])
629 def test_strip(self, a, chars, out, dt):
630 a = np.array(a, dtype=dt)
631 if chars is not None:
632 chars = np.array(chars, dtype=dt)
633 out = np.array(out, dtype=dt)
634 assert_array_equal(np.strings.strip(a, chars), out)
635
636 @pytest.mark.parametrize("buf,old,new,count,res", [
637 ("", "", "", -1, ""),
638 ("", "", "A", -1, "A"),
639 ("", "A", "", -1, ""),
640 ("", "A", "A", -1, ""),
641 ("", "", "", 100, ""),
642 ("", "", "A", 100, "A"),
643 ("A", "", "", -1, "A"),
644 ("A", "", "*", -1, "*A*"),
645 ("A", "", "*1", -1, "*1A*1"),
646 ("A", "", "*-#", -1, "*-#A*-#"),
647 ("AA", "", "*-", -1, "*-A*-A*-"),
648 ("AA", "", "*-", -1, "*-A*-A*-"),
649 ("AA", "", "*-", 4, "*-A*-A*-"),
650 ("AA", "", "*-", 3, "*-A*-A*-"),
651 ("AA", "", "*-", 2, "*-A*-A"),
652 ("AA", "", "*-", 1, "*-AA"),
653 ("AA", "", "*-", 0, "AA"),
654 ("A", "A", "", -1, ""),
655 ("AAA", "A", "", -1, ""),
656 ("AAA", "A", "", -1, ""),
657 ("AAA", "A", "", 4, ""),
658 ("AAA", "A", "", 3, ""),
659 ("AAA", "A", "", 2, "A"),
660 ("AAA", "A", "", 1, "AA"),
661 ("AAA", "A", "", 0, "AAA"),
662 ("AAAAAAAAAA", "A", "", -1, ""),
663 ("ABACADA", "A", "", -1, "BCD"),
664 ("ABACADA", "A", "", -1, "BCD"),
665 ("ABACADA", "A", "", 5, "BCD"),
666 ("ABACADA", "A", "", 4, "BCD"),
667 ("ABACADA", "A", "", 3, "BCDA"),
668 ("ABACADA", "A", "", 2, "BCADA"),
669 ("ABACADA", "A", "", 1, "BACADA"),
670 ("ABACADA", "A", "", 0, "ABACADA"),
671 ("ABCAD", "A", "", -1, "BCD"),
672 ("ABCADAA", "A", "", -1, "BCD"),
673 ("BCD", "A", "", -1, "BCD"),
674 ("*************", "A", "", -1, "*************"),
675 ("^" + "A" * 1000 + "^", "A", "", 999, "^A^"),
676 ("the", "the", "", -1, ""),
677 ("theater", "the", "", -1, "ater"),
678 ("thethe", "the", "", -1, ""),
679 ("thethethethe", "the", "", -1, ""),
680 ("theatheatheathea", "the", "", -1, "aaaa"),
681 ("that", "the", "", -1, "that"),
682 ("thaet", "the", "", -1, "thaet"),
683 ("here and there", "the", "", -1, "here and re"),
684 ("here and there and there", "the", "", -1, "here and re and re"),
685 ("here and there and there", "the", "", 3, "here and re and re"),
686 ("here and there and there", "the", "", 2, "here and re and re"),
687 ("here and there and there", "the", "", 1, "here and re and there"),
688 ("here and there and there", "the", "", 0, "here and there and there"),
689 ("here and there and there", "the", "", -1, "here and re and re"),
690 ("abc", "the", "", -1, "abc"),
691 ("abcdefg", "the", "", -1, "abcdefg"),
692 ("bbobob", "bob", "", -1, "bob"),
693 ("bbobobXbbobob", "bob", "", -1, "bobXbob"),
694 ("aaaaaaabob", "bob", "", -1, "aaaaaaa"),
695 ("aaaaaaa", "bob", "", -1, "aaaaaaa"),
696 ("Who goes there?", "o", "o", -1, "Who goes there?"),
697 ("Who goes there?", "o", "O", -1, "WhO gOes there?"),
698 ("Who goes there?", "o", "O", -1, "WhO gOes there?"),
699 ("Who goes there?", "o", "O", 3, "WhO gOes there?"),
700 ("Who goes there?", "o", "O", 2, "WhO gOes there?"),
701 ("Who goes there?", "o", "O", 1, "WhO goes there?"),
702 ("Who goes there?", "o", "O", 0, "Who goes there?"),
703 ("Who goes there?", "a", "q", -1, "Who goes there?"),
704 ("Who goes there?", "W", "w", -1, "who goes there?"),
705 ("WWho goes there?WW", "W", "w", -1, "wwho goes there?ww"),
706 ("Who goes there?", "?", "!", -1, "Who goes there!"),
707 ("Who goes there??", "?", "!", -1, "Who goes there!!"),
708 ("Who goes there?", ".", "!", -1, "Who goes there?"),
709 ("This is a tissue", "is", "**", -1, "Th** ** a t**sue"),
710 ("This is a tissue", "is", "**", -1, "Th** ** a t**sue"),
711 ("This is a tissue", "is", "**", 4, "Th** ** a t**sue"),
712 ("This is a tissue", "is", "**", 3, "Th** ** a t**sue"),
713 ("This is a tissue", "is", "**", 2, "Th** ** a tissue"),
714 ("This is a tissue", "is", "**", 1, "Th** is a tissue"),
715 ("This is a tissue", "is", "**", 0, "This is a tissue"),
716 ("bobob", "bob", "cob", -1, "cobob"),
717 ("bobobXbobobob", "bob", "cob", -1, "cobobXcobocob"),
718 ("bobob", "bot", "bot", -1, "bobob"),
719 ("Reykjavik", "k", "KK", -1, "ReyKKjaviKK"),
720 ("Reykjavik", "k", "KK", -1, "ReyKKjaviKK"),
721 ("Reykjavik", "k", "KK", 2, "ReyKKjaviKK"),
722 ("Reykjavik", "k", "KK", 1, "ReyKKjavik"),
723 ("Reykjavik", "k", "KK", 0, "Reykjavik"),
724 ("A.B.C.", ".", "----", -1, "A----B----C----"),
725 ("Reykjavik", "q", "KK", -1, "Reykjavik"),
726 ("spam, spam, eggs and spam", "spam", "ham", -1,
727 "ham, ham, eggs and ham"),
728 ("spam, spam, eggs and spam", "spam", "ham", -1,
729 "ham, ham, eggs and ham"),
730 ("spam, spam, eggs and spam", "spam", "ham", 4,
731 "ham, ham, eggs and ham"),
732 ("spam, spam, eggs and spam", "spam", "ham", 3,
733 "ham, ham, eggs and ham"),
734 ("spam, spam, eggs and spam", "spam", "ham", 2,
735 "ham, ham, eggs and spam"),
736 ("spam, spam, eggs and spam", "spam", "ham", 1,
737 "ham, spam, eggs and spam"),
738 ("spam, spam, eggs and spam", "spam", "ham", 0,
739 "spam, spam, eggs and spam"),
740 ("bobobob", "bobob", "bob", -1, "bobob"),
741 ("bobobobXbobobob", "bobob", "bob", -1, "bobobXbobob"),
742 ("BOBOBOB", "bob", "bobby", -1, "BOBOBOB"),
743 ("one!two!three!", "!", "@", 1, "one@two!three!"),
744 ("one!two!three!", "!", "", -1, "onetwothree"),
745 ("one!two!three!", "!", "@", 2, "one@two@three!"),
746 ("one!two!three!", "!", "@", 3, "one@two@three@"),
747 ("one!two!three!", "!", "@", 4, "one@two@three@"),
748 ("one!two!three!", "!", "@", 0, "one!two!three!"),
749 ("one!two!three!", "!", "@", -1, "one@two@three@"),
750 ("one!two!three!", "x", "@", -1, "one!two!three!"),
751 ("one!two!three!", "x", "@", 2, "one!two!three!"),
752 ("abc", "", "-", -1, "-a-b-c-"),
753 ("abc", "", "-", 3, "-a-b-c"),
754 ("abc", "", "-", 0, "abc"),
755 ("abc", "ab", "--", 0, "abc"),
756 ("abc", "xy", "--", -1, "abc"),
757 (["abbc", "abbd"], "b", "z", [1, 2], ["azbc", "azzd"]),
758 ])
759 def test_replace(self, buf, old, new, count, res, dt):
760 if "😊" in buf and dt == "S":
761 pytest.skip("Bytes dtype does not support non-ascii input")
762 buf = np.array(buf, dtype=dt)
763 old = np.array(old, dtype=dt)
764 new = np.array(new, dtype=dt)
765 res = np.array(res, dtype=dt)
766 assert_array_equal(np.strings.replace(buf, old, new, count), res)
767
768 @pytest.mark.parametrize("buf,sub,start,end,res", [
769 ("abcdefghiabc", "", 0, None, 0),
770 ("abcdefghiabc", "def", 0, None, 3),
771 ("abcdefghiabc", "abc", 0, None, 0),
772 ("abcdefghiabc", "abc", 1, None, 9),
773 ])
774 def test_index(self, buf, sub, start, end, res, dt):
775 buf = np.array(buf, dtype=dt)
776 sub = np.array(sub, dtype=dt)
777 assert_array_equal(np.strings.index(buf, sub, start, end), res)
778
779 @pytest.mark.parametrize("buf,sub,start,end", [
780 ("abcdefghiabc", "hib", 0, None),
781 ("abcdefghiab", "abc", 1, None),
782 ("abcdefghi", "ghi", 8, None),
783 ("abcdefghi", "ghi", -1, None),
784 ("rrarrrrrrrrra", "a", 4, 6),
785 ])
786 def test_index_raises(self, buf, sub, start, end, dt):
787 buf = np.array(buf, dtype=dt)
788 sub = np.array(sub, dtype=dt)
789 with pytest.raises(ValueError, match="substring not found"):
790 np.strings.index(buf, sub, start, end)
791
792 @pytest.mark.parametrize("buf,sub,start,end,res", [
793 ("abcdefghiabc", "", 0, None, 12),
794 ("abcdefghiabc", "def", 0, None, 3),
795 ("abcdefghiabc", "abc", 0, None, 9),
796 ("abcdefghiabc", "abc", 0, -1, 0),
797 ])
798 def test_rindex(self, buf, sub, start, end, res, dt):
799 buf = np.array(buf, dtype=dt)
800 sub = np.array(sub, dtype=dt)
801 assert_array_equal(np.strings.rindex(buf, sub, start, end), res)
802
803 @pytest.mark.parametrize("buf,sub,start,end", [
804 ("abcdefghiabc", "hib", 0, None),
805 ("defghiabc", "def", 1, None),
806 ("defghiabc", "abc", 0, -1),
807 ("abcdefghi", "ghi", 0, 8),
808 ("abcdefghi", "ghi", 0, -1),
809 ("rrarrrrrrrrra", "a", 4, 6),
810 ])
811 def test_rindex_raises(self, buf, sub, start, end, dt):
812 buf = np.array(buf, dtype=dt)
813 sub = np.array(sub, dtype=dt)
814 with pytest.raises(ValueError, match="substring not found"):
815 np.strings.rindex(buf, sub, start, end)
816
817 @pytest.mark.parametrize("buf,tabsize,res", [
818 ("abc\rab\tdef\ng\thi", 8, "abc\rab def\ng hi"),
819 ("abc\rab\tdef\ng\thi", 4, "abc\rab def\ng hi"),
820 ("abc\r\nab\tdef\ng\thi", 8, "abc\r\nab def\ng hi"),
821 ("abc\r\nab\tdef\ng\thi", 4, "abc\r\nab def\ng hi"),
822 ("abc\r\nab\r\ndef\ng\r\nhi", 4, "abc\r\nab\r\ndef\ng\r\nhi"),
823 (" \ta\n\tb", 1, " a\n b"),
824 ])
825 def test_expandtabs(self, buf, tabsize, res, dt):
826 buf = np.array(buf, dtype=dt)
827 res = np.array(res, dtype=dt)
828 assert_array_equal(np.strings.expandtabs(buf, tabsize), res)
829
830 def test_expandtabs_raises_overflow(self, dt):
831 with pytest.raises(OverflowError, match="new string is too long"):
832 np.strings.expandtabs(np.array("\ta\n\tb", dtype=dt), sys.maxsize)
833 np.strings.expandtabs(np.array("\ta\n\tb", dtype=dt), 2**61)
834
835 def test_expandtabs_length_not_cause_segfault(self, dt):
836 # see gh-28829
837 with pytest.raises(
838 _UFuncNoLoopError,
839 match="did not contain a loop with signature matching types",
840 ):
841 np._core.strings._expandtabs_length.reduce(np.zeros(200))
842
843 with pytest.raises(
844 _UFuncNoLoopError,
845 match="did not contain a loop with signature matching types",
846 ):
847 np.strings.expandtabs(np.zeros(200))
848
849 FILL_ERROR = "The fill character must be exactly one character long"
850
851 def test_center_raises_multiple_character_fill(self, dt):
852 buf = np.array("abc", dtype=dt)
853 fill = np.array("**", dtype=dt)
854 with pytest.raises(TypeError, match=self.FILL_ERROR):
855 np.strings.center(buf, 10, fill)
856
857 def test_ljust_raises_multiple_character_fill(self, dt):
858 buf = np.array("abc", dtype=dt)
859 fill = np.array("**", dtype=dt)
860 with pytest.raises(TypeError, match=self.FILL_ERROR):
861 np.strings.ljust(buf, 10, fill)
862
863 def test_rjust_raises_multiple_character_fill(self, dt):
864 buf = np.array("abc", dtype=dt)
865 fill = np.array("**", dtype=dt)
866 with pytest.raises(TypeError, match=self.FILL_ERROR):
867 np.strings.rjust(buf, 10, fill)
868
869 @pytest.mark.parametrize("buf,width,fillchar,res", [
870 ('abc', 10, ' ', ' abc '),
871 ('abc', 6, ' ', ' abc '),
872 ('abc', 3, ' ', 'abc'),
873 ('abc', 2, ' ', 'abc'),
874 ('abc', -2, ' ', 'abc'),
875 ('abc', 10, '*', '***abc****'),
876 ])
877 def test_center(self, buf, width, fillchar, res, dt):
878 buf = np.array(buf, dtype=dt)
879 fillchar = np.array(fillchar, dtype=dt)
880 res = np.array(res, dtype=dt)
881 assert_array_equal(np.strings.center(buf, width, fillchar), res)
882
883 @pytest.mark.parametrize("buf,width,fillchar,res", [
884 ('abc', 10, ' ', 'abc '),
885 ('abc', 6, ' ', 'abc '),
886 ('abc', 3, ' ', 'abc'),
887 ('abc', 2, ' ', 'abc'),
888 ('abc', -2, ' ', 'abc'),
889 ('abc', 10, '*', 'abc*******'),
890 ])
891 def test_ljust(self, buf, width, fillchar, res, dt):
892 buf = np.array(buf, dtype=dt)
893 fillchar = np.array(fillchar, dtype=dt)
894 res = np.array(res, dtype=dt)
895 assert_array_equal(np.strings.ljust(buf, width, fillchar), res)
896
897 @pytest.mark.parametrize("buf,width,fillchar,res", [
898 ('abc', 10, ' ', ' abc'),
899 ('abc', 6, ' ', ' abc'),
900 ('abc', 3, ' ', 'abc'),
901 ('abc', 2, ' ', 'abc'),
902 ('abc', -2, ' ', 'abc'),
903 ('abc', 10, '*', '*******abc'),
904 ])
905 def test_rjust(self, buf, width, fillchar, res, dt):
906 buf = np.array(buf, dtype=dt)
907 fillchar = np.array(fillchar, dtype=dt)
908 res = np.array(res, dtype=dt)
909 assert_array_equal(np.strings.rjust(buf, width, fillchar), res)
910
911 @pytest.mark.parametrize("buf,width,res", [
912 ('123', 2, '123'),
913 ('123', 3, '123'),
914 ('0123', 4, '0123'),
915 ('+123', 3, '+123'),
916 ('+123', 4, '+123'),
917 ('+123', 5, '+0123'),
918 ('+0123', 5, '+0123'),
919 ('-123', 3, '-123'),
920 ('-123', 4, '-123'),
921 ('-0123', 5, '-0123'),
922 ('000', 3, '000'),
923 ('34', 1, '34'),
924 ('34', -1, '34'),
925 ('0034', 4, '0034'),
926 ])
927 def test_zfill(self, buf, width, res, dt):
928 buf = np.array(buf, dtype=dt)
929 res = np.array(res, dtype=dt)
930 assert_array_equal(np.strings.zfill(buf, width), res)
931
932 @pytest.mark.parametrize("buf,sep,res1,res2,res3", [
933 ("this is the partition method", "ti", "this is the par",
934 "ti", "tion method"),
935 ("http://www.python.org", "://", "http", "://", "www.python.org"),
936 ("http://www.python.org", "?", "http://www.python.org", "", ""),
937 ("http://www.python.org", "http://", "", "http://", "www.python.org"),
938 ("http://www.python.org", "org", "http://www.python.", "org", ""),
939 ("http://www.python.org", ["://", "?", "http://", "org"],
940 ["http", "http://www.python.org", "", "http://www.python."],
941 ["://", "", "http://", "org"],
942 ["www.python.org", "", "www.python.org", ""]),
943 ("mississippi", "ss", "mi", "ss", "issippi"),
944 ("mississippi", "i", "m", "i", "ssissippi"),
945 ("mississippi", "w", "mississippi", "", ""),
946 ])
947 def test_partition(self, buf, sep, res1, res2, res3, dt):
948 buf = np.array(buf, dtype=dt)
949 sep = np.array(sep, dtype=dt)
950 res1 = np.array(res1, dtype=dt)
951 res2 = np.array(res2, dtype=dt)
952 res3 = np.array(res3, dtype=dt)
953 act1, act2, act3 = np.strings.partition(buf, sep)
954 assert_array_equal(act1, res1)
955 assert_array_equal(act2, res2)
956 assert_array_equal(act3, res3)
957 assert_array_equal(act1 + act2 + act3, buf)
958
959 @pytest.mark.parametrize("buf,sep,res1,res2,res3", [
960 ("this is the partition method", "ti", "this is the parti",
961 "ti", "on method"),
962 ("http://www.python.org", "://", "http", "://", "www.python.org"),
963 ("http://www.python.org", "?", "", "", "http://www.python.org"),
964 ("http://www.python.org", "http://", "", "http://", "www.python.org"),
965 ("http://www.python.org", "org", "http://www.python.", "org", ""),
966 ("http://www.python.org", ["://", "?", "http://", "org"],
967 ["http", "", "", "http://www.python."],
968 ["://", "", "http://", "org"],
969 ["www.python.org", "http://www.python.org", "www.python.org", ""]),
970 ("mississippi", "ss", "missi", "ss", "ippi"),
971 ("mississippi", "i", "mississipp", "i", ""),
972 ("mississippi", "w", "", "", "mississippi"),
973 ])
974 def test_rpartition(self, buf, sep, res1, res2, res3, dt):
975 buf = np.array(buf, dtype=dt)
976 sep = np.array(sep, dtype=dt)
977 res1 = np.array(res1, dtype=dt)
978 res2 = np.array(res2, dtype=dt)
979 res3 = np.array(res3, dtype=dt)
980 act1, act2, act3 = np.strings.rpartition(buf, sep)
981 assert_array_equal(act1, res1)
982 assert_array_equal(act2, res2)
983 assert_array_equal(act3, res3)
984 assert_array_equal(act1 + act2 + act3, buf)
985
986 @pytest.mark.parametrize("args", [
987 (None,),
988 (None, None),
989 (None, None, -1),
990 (0,),
991 (0, None),
992 (0, None, -1),
993 (1,),
994 (1, None),
995 (1, None, -1),
996 (3,),
997 (3, None),
998 (5,),
999 (5, None),
1000 (5, 5),
1001 (5, 5, -1),
1002 (6,), # test index past the end
1003 (6, None),
1004 (6, None, -1),
1005 (6, 7), # test start and stop index past the end
1006 (4, 3), # test start > stop index
1007 (-1,),
1008 (-1, None),
1009 (-1, None, -1),
1010 (-3,),
1011 (-3, None),
1012 ([3, 4],),
1013 ([3, 4], None),
1014 ([2, 4],),
1015 ([-3, 5],),
1016 ([-3, 5], None),
1017 ([-3, 5], None, -1),
1018 ([0, -5],),
1019 ([0, -5], None),
1020 ([0, -5], None, -1),
1021 (1, 4),
1022 (-3, 5),
1023 (None, -1),
1024 (0, [4, 2]),
1025 ([1, 2], [-1, -2]),
1026 (1, 5, 2),
1027 (None, None, -1),
1028 ([0, 6], [-1, 0], [2, -1]),
1029 ])
1030 @pytest.mark.parametrize("buf", [
1031 ["hello", "world"],
1032 ['hello world', 'γεια σου κόσμε', '你好世界', '👋 🌍'],
1033 ])
1034 def test_slice(self, args, buf, dt):
1035 if dt == "S" and "你好世界" in buf:
1036 pytest.skip("Bytes dtype does not support non-ascii input")
1037 if len(buf) == 4:
1038 args = tuple(s * 2 if isinstance(s, list) else s for s in args)
1039 buf = np.array(buf, dtype=dt)
1040 act = np.strings.slice(buf, *args)
1041 bcast_args = tuple(np.broadcast_to(arg, buf.shape) for arg in args)
1042 res = np.array([s[slice(*arg)]
1043 for s, arg in zip(buf, zip(*bcast_args))],
1044 dtype=dt)
1045 assert_array_equal(act, res)
1046
1047 def test_slice_unsupported(self, dt):
1048 with pytest.raises(TypeError, match="did not contain a loop"):
1049 np.strings.slice(np.array([1, 2, 3]), 4)
1050
1051 regexp = (r"Cannot cast ufunc '_slice' input .* "
1052 r"from .* to dtype\('int(64|32)'\)")
1053 with pytest.raises(TypeError, match=regexp):
1054 np.strings.slice(np.array(['foo', 'bar'], dtype=dt),
1055 np.array(['foo', 'bar'], dtype=dt))
1056
1057 @pytest.mark.parametrize("int_dt", [np.int8, np.int16, np.int32,
1058 np.int64, np.uint8, np.uint16,
1059 np.uint32, np.uint64])
1060 def test_slice_int_type_promotion(self, int_dt, dt):
1061 buf = np.array(["hello", "world"], dtype=dt)
1062 np_slice = np.strings.slice
1063 assert_array_equal(np_slice(buf, int_dt(4)),
1064 np.array(["hell", "worl"], dtype=dt))
1065 assert_array_equal(np_slice(buf, np.array([4, 4], dtype=int_dt)),
1066 np.array(["hell", "worl"], dtype=dt))
1067
1068 assert_array_equal(np_slice(buf, int_dt(2), int_dt(4)),
1069 np.array(["ll", "rl"], dtype=dt))
1070 assert_array_equal(np_slice(buf, np.array([2, 2], dtype=int_dt),
1071 np.array([4, 4], dtype=int_dt)),
1072 np.array(["ll", "rl"], dtype=dt))
1073
1074 assert_array_equal(np_slice(buf, int_dt(0), int_dt(4), int_dt(2)),
1075 np.array(["hl", "wr"], dtype=dt))
1076 assert_array_equal(np_slice(buf,
1077 np.array([0, 0], dtype=int_dt),
1078 np.array([4, 4], dtype=int_dt),
1079 np.array([2, 2], dtype=int_dt)),
1080 np.array(["hl", "wr"], dtype=dt))
1081
1082@pytest.mark.parametrize("dt", ["U", "T"])
1083class TestMethodsWithUnicode:
1084 @pytest.mark.parametrize("in_,out", [
1085 ("", False),
1086 ("a", False),
1087 ("0", True),
1088 ("\u2460", False), # CIRCLED DIGIT 1
1089 ("\xbc", False), # VULGAR FRACTION ONE QUARTER
1090 ("\u0660", True), # ARABIC_INDIC DIGIT ZERO
1091 ("012345", True),
1092 ("012345a", False),
1093 (["0", "a"], [True, False]),
1094 ])
1095 def test_isdecimal_unicode(self, in_, out, dt):
1096 buf = np.array(in_, dtype=dt)
1097 assert_array_equal(np.strings.isdecimal(buf), out)
1098
1099 @pytest.mark.parametrize("in_,out", [
1100 ("", False),
1101 ("a", False),
1102 ("0", True),
1103 ("\u2460", True), # CIRCLED DIGIT 1
1104 ("\xbc", True), # VULGAR FRACTION ONE QUARTER
1105 ("\u0660", True), # ARABIC_INDIC DIGIT ZERO
1106 ("012345", True),
1107 ("012345a", False),
1108 (["0", "a"], [True, False]),
1109 ])
1110 def test_isnumeric_unicode(self, in_, out, dt):
1111 buf = np.array(in_, dtype=dt)
1112 assert_array_equal(np.strings.isnumeric(buf), out)
1113
1114 @pytest.mark.parametrize("buf,old,new,count,res", [
1115 ("...\u043c......<", "<", "<", -1, "...\u043c......<"),
1116 ("Ae¢☃€ 😊" * 2, "A", "B", -1, "Be¢☃€ 😊Be¢☃€ 😊"),
1117 ("Ae¢☃€ 😊" * 2, "😊", "B", -1, "Ae¢☃€ BAe¢☃€ B"),
1118 ])
1119 def test_replace_unicode(self, buf, old, new, count, res, dt):
1120 buf = np.array(buf, dtype=dt)
1121 old = np.array(old, dtype=dt)
1122 new = np.array(new, dtype=dt)
1123 res = np.array(res, dtype=dt)
1124 assert_array_equal(np.strings.replace(buf, old, new, count), res)
1125
1126 @pytest.mark.parametrize("in_", [
1127 '\U00010401',
1128 '\U00010427',
1129 '\U00010429',
1130 '\U0001044E',
1131 '\U0001D7F6',
1132 '\U00011066',
1133 '\U000104A0',
1134 pytest.param('\U0001F107', marks=pytest.mark.xfail(
1135 sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1136 reason="PYPY bug in Py_UNICODE_ISALNUM",
1137 strict=True)),
1138 ])
1139 def test_isalnum_unicode(self, in_, dt):
1140 in_ = np.array(in_, dtype=dt)
1141 assert_array_equal(np.strings.isalnum(in_), True)
1142
1143 @pytest.mark.parametrize("in_,out", [
1144 ('\u1FFc', False),
1145 ('\u2167', False),
1146 ('\U00010401', False),
1147 ('\U00010427', False),
1148 ('\U0001F40D', False),
1149 ('\U0001F46F', False),
1150 ('\u2177', True),
1151 pytest.param('\U00010429', True, marks=pytest.mark.xfail(
1152 sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1153 reason="PYPY bug in Py_UNICODE_ISLOWER",
1154 strict=True)),
1155 ('\U0001044E', True),
1156 ])
1157 def test_islower_unicode(self, in_, out, dt):
1158 in_ = np.array(in_, dtype=dt)
1159 assert_array_equal(np.strings.islower(in_), out)
1160
1161 @pytest.mark.parametrize("in_,out", [
1162 ('\u1FFc', False),
1163 ('\u2167', True),
1164 ('\U00010401', True),
1165 ('\U00010427', True),
1166 ('\U0001F40D', False),
1167 ('\U0001F46F', False),
1168 ('\u2177', False),
1169 pytest.param('\U00010429', False, marks=pytest.mark.xfail(
1170 sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1171 reason="PYPY bug in Py_UNICODE_ISUPPER",
1172 strict=True)),
1173 ('\U0001044E', False),
1174 ])
1175 def test_isupper_unicode(self, in_, out, dt):
1176 in_ = np.array(in_, dtype=dt)
1177 assert_array_equal(np.strings.isupper(in_), out)
1178
1179 @pytest.mark.parametrize("in_,out", [
1180 ('\u1FFc', True),
1181 ('Greek \u1FFcitlecases ...', True),
1182 pytest.param('\U00010401\U00010429', True, marks=pytest.mark.xfail(
1183 sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1184 reason="PYPY bug in Py_UNICODE_ISISTITLE",
1185 strict=True)),
1186 ('\U00010427\U0001044E', True),
1187 pytest.param('\U00010429', False, marks=pytest.mark.xfail(
1188 sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1189 reason="PYPY bug in Py_UNICODE_ISISTITLE",
1190 strict=True)),
1191 ('\U0001044E', False),
1192 ('\U0001F40D', False),
1193 ('\U0001F46F', False),
1194 ])
1195 def test_istitle_unicode(self, in_, out, dt):
1196 in_ = np.array(in_, dtype=dt)
1197 assert_array_equal(np.strings.istitle(in_), out)
1198
1199 @pytest.mark.parametrize("buf,sub,start,end,res", [
1200 ("Ae¢☃€ 😊" * 2, "😊", 0, None, 6),
