Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
test_strings.py1524 linesDownload Raw Back to tests
1import operator
2import sys
3
4import pytest
5
6import numpy as np
7from numpy._core._exceptions import _UFuncNoLoopError
8from numpy.testing import IS_PYPY, assert_array_equal, assert_raises
9from numpy.testing._private.utils import requires_memory
10
11COMPARISONS = [
12    (operator.eq, np.equal, "=="),
13    (operator.ne, np.not_equal, "!="),
14    (operator.lt, np.less, "<"),
15    (operator.le, np.less_equal, "<="),
16    (operator.gt, np.greater, ">"),
17    (operator.ge, np.greater_equal, ">="),
18]
19
20MAX = np.iinfo(np.int64).max
21
22IS_PYPY_LT_7_3_16 = IS_PYPY and sys.implementation.version < (7, 3, 16)
23
24@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
25def test_mixed_string_comparison_ufuncs_fail(op, ufunc, sym):
26    arr_string = np.array(["a", "b"], dtype="S")
27    arr_unicode = np.array(["a", "c"], dtype="U")
28
29    with pytest.raises(TypeError, match="did not contain a loop"):
30        ufunc(arr_string, arr_unicode)
31
32    with pytest.raises(TypeError, match="did not contain a loop"):
33        ufunc(arr_unicode, arr_string)
34
35@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
36def test_mixed_string_comparisons_ufuncs_with_cast(op, ufunc, sym):
37    arr_string = np.array(["a", "b"], dtype="S")
38    arr_unicode = np.array(["a", "c"], dtype="U")
39
40    # While there is no loop, manual casting is acceptable:
41    res1 = ufunc(arr_string, arr_unicode, signature="UU->?", casting="unsafe")
42    res2 = ufunc(arr_string, arr_unicode, signature="SS->?", casting="unsafe")
43
44    expected = op(arr_string.astype("U"), arr_unicode)
45    assert_array_equal(res1, expected)
46    assert_array_equal(res2, expected)
47
48
49@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
50@pytest.mark.parametrize("dtypes", [
51        ("S2", "S2"), ("S2", "S10"),
52        ("<U1", "<U1"), ("<U1", ">U1"), (">U1", ">U1"),
53        ("<U1", "<U10"), ("<U1", ">U10")])
54@pytest.mark.parametrize("aligned", [True, False])
55def test_string_comparisons(op, ufunc, sym, dtypes, aligned):
56    # ensure native byte-order for the first view to stay within unicode range
57    native_dt = np.dtype(dtypes[0]).newbyteorder("=")
58    arr = np.arange(2**15).view(native_dt).astype(dtypes[0])
59    if not aligned:
60        # Make `arr` unaligned:
61        new = np.zeros(arr.nbytes + 1, dtype=np.uint8)[1:].view(dtypes[0])
62        new[...] = arr
63        arr = new
64
65    arr2 = arr.astype(dtypes[1], copy=True)
66    np.random.shuffle(arr2)
67    arr[0] = arr2[0]  # make sure one matches
68
69    expected = [op(d1, d2) for d1, d2 in zip(arr.tolist(), arr2.tolist())]
70    assert_array_equal(op(arr, arr2), expected)
71    assert_array_equal(ufunc(arr, arr2), expected)
72    assert_array_equal(
73        np.char.compare_chararrays(arr, arr2, sym, False), expected
74    )
75
76    expected = [op(d2, d1) for d1, d2 in zip(arr.tolist(), arr2.tolist())]
77    assert_array_equal(op(arr2, arr), expected)
78    assert_array_equal(ufunc(arr2, arr), expected)
79    assert_array_equal(
80        np.char.compare_chararrays(arr2, arr, sym, False), expected
81    )
82
83
84@pytest.mark.parametrize(["op", "ufunc", "sym"], COMPARISONS)
85@pytest.mark.parametrize("dtypes", [
86        ("S2", "S2"), ("S2", "S10"), ("<U1", "<U1"), ("<U1", ">U10")])
87def test_string_comparisons_empty(op, ufunc, sym, dtypes):
88    arr = np.empty((1, 0, 1, 5), dtype=dtypes[0])
89    arr2 = np.empty((100, 1, 0, 1), dtype=dtypes[1])
90
91    expected = np.empty(np.broadcast_shapes(arr.shape, arr2.shape), dtype=bool)
92    assert_array_equal(op(arr, arr2), expected)
93    assert_array_equal(ufunc(arr, arr2), expected)
94    assert_array_equal(
95        np.char.compare_chararrays(arr, arr2, sym, False), expected
96    )
97
98
99@pytest.mark.parametrize("str_dt", ["S", "U"])
100@pytest.mark.parametrize("float_dt", np.typecodes["AllFloat"])
101def test_float_to_string_cast(str_dt, float_dt):
102    float_dt = np.dtype(float_dt)
103    fi = np.finfo(float_dt)
104    arr = np.array([np.nan, np.inf, -np.inf, fi.max, fi.min], dtype=float_dt)
105    expected = ["nan", "inf", "-inf", str(fi.max), str(fi.min)]
106    if float_dt.kind == "c":
107        expected = [f"({r}+0j)" for r in expected]
108
109    res = arr.astype(str_dt)
110    assert_array_equal(res, np.array(expected, dtype=str_dt))
111
112
113@pytest.mark.parametrize("str_dt", "US")
114@pytest.mark.parametrize("size", [-1, np.iinfo(np.intc).max])
115def test_string_size_dtype_errors(str_dt, size):
116    if size > 0:
117        size = size // np.dtype(f"{str_dt}1").itemsize + 1
118
119    with pytest.raises(ValueError):
120        np.dtype((str_dt, size))
121    with pytest.raises(TypeError):
122        np.dtype(f"{str_dt}{size}")
123
124
125@pytest.mark.parametrize("str_dt", "US")
126def test_string_size_dtype_large_repr(str_dt):
127    size = np.iinfo(np.intc).max // np.dtype(f"{str_dt}1").itemsize
128    size_str = str(size)
129
130    dtype = np.dtype((str_dt, size))
131    assert size_str in dtype.str
132    assert size_str in str(dtype)
133    assert size_str in repr(dtype)
134
135
136@pytest.mark.slow
137@requires_memory(2 * np.iinfo(np.intc).max)
138@pytest.mark.parametrize("str_dt", "US")
139@pytest.mark.thread_unsafe(reason="crashes with low memory")
140def test_large_string_coercion_error(str_dt):
141    very_large = np.iinfo(np.intc).max // np.dtype(f"{str_dt}1").itemsize
142    try:
143        large_string = "A" * (very_large + 1)
144    except Exception:
145        # We may not be able to create this Python string on 32bit.
146        pytest.skip("python failed to create huge string")
147
148    class MyStr:
149        def __str__(self):
150            return large_string
151
152    try:
153        # TypeError from NumPy, or OverflowError from 32bit Python.
154        with pytest.raises((TypeError, OverflowError)):
155            np.array([large_string], dtype=str_dt)
156
157        # Same as above, but input has to be converted to a string.
158        with pytest.raises((TypeError, OverflowError)):
159            np.array([MyStr()], dtype=str_dt)
160    except MemoryError:
161        # Catch memory errors, because `requires_memory` would do so.
162        raise AssertionError("Ops should raise before any large allocation.")
163
164@pytest.mark.slow
165@requires_memory(2 * np.iinfo(np.intc).max)
166@pytest.mark.parametrize("str_dt", "US")
167@pytest.mark.thread_unsafe(reason="crashes with low memory")
168def test_large_string_addition_error(str_dt):
169    very_large = np.iinfo(np.intc).max // np.dtype(f"{str_dt}1").itemsize
170
171    a = np.array(["A" * very_large], dtype=str_dt)
172    b = np.array("B", dtype=str_dt)
173    try:
174        with pytest.raises(TypeError):
175            np.add(a, b)
176        with pytest.raises(TypeError):
177            np.add(a, a)
178    except MemoryError:
179        # Catch memory errors, because `requires_memory` would do so.
180        raise AssertionError("Ops should raise before any large allocation.")
181
182
183def test_large_string_cast():
184    very_large = np.iinfo(np.intc).max // 4
185    # Could be nice to test very large path, but it makes too many huge
186    # allocations right now (need non-legacy cast loops for this).
187    # a = np.array([], dtype=np.dtype(("S", very_large)))
188    # assert a.astype("U").dtype.itemsize == very_large * 4
189
190    a = np.array([], dtype=np.dtype(("S", very_large + 1)))
191    # It is not perfect but OK if this raises a MemoryError during setup
192    # (this happens due clunky code and/or buffer setup.)
193    with pytest.raises((TypeError, MemoryError)):
194        a.astype("U")
195
196
197@pytest.mark.parametrize("dt", ["S1", "U1"])
198def test_in_place_mutiply_no_overflow(dt):
199    # see gh-30495
200    a = np.array("a", dtype=dt)
201    a *= 20
202    assert_array_equal(a, np.array("a", dtype=dt))
203
204
205@pytest.mark.parametrize("dt", ["S", "U", "T"])
206class TestMethods:
207
208    @pytest.mark.parametrize("in1,in2,out", [
209        ("", "", ""),
210        ("abc", "abc", "abcabc"),
211        ("12345", "12345", "1234512345"),
212        ("MixedCase", "MixedCase", "MixedCaseMixedCase"),
213        ("12345 \0 ", "12345 \0 ", "12345 \0 12345 \0 "),
214        ("UPPER", "UPPER", "UPPERUPPER"),
215        (["abc", "def"], ["hello", "world"], ["abchello", "defworld"]),
216    ])
217    def test_add(self, in1, in2, out, dt):
218        in1 = np.array(in1, dtype=dt)
219        in2 = np.array(in2, dtype=dt)
220        out = np.array(out, dtype=dt)
221        assert_array_equal(np.strings.add(in1, in2), out)
222
223    @pytest.mark.parametrize("in1,in2,out", [
224        ("abc", 3, "abcabcabc"),
225        ("abc", 0, ""),
226        ("abc", -1, ""),
227        (["abc", "def"], [1, 4], ["abc", "defdefdefdef"]),
228    ])
229    def test_multiply(self, in1, in2, out, dt):
230        in1 = np.array(in1, dtype=dt)
231        out = np.array(out, dtype=dt)
232        assert_array_equal(np.strings.multiply(in1, in2), out)
233
234    def test_multiply_raises(self, dt):
235        with pytest.raises(TypeError, match="unsupported type"):
236            np.strings.multiply(np.array("abc", dtype=dt), 3.14)
237
238        with pytest.raises(OverflowError):
239            np.strings.multiply(np.array("abc", dtype=dt), sys.maxsize)
240
241    def test_inplace_multiply(self, dt):
242        arr = np.array(['foo ', 'bar'], dtype=dt)
243        arr *= 2
244        if dt != "T":
245            assert_array_equal(arr, np.array(['foo ', 'barb'], dtype=dt))
246        else:
247            assert_array_equal(arr, ['foo foo ', 'barbar'])
248
249        with pytest.raises(OverflowError):
250            arr *= sys.maxsize
251
252    @pytest.mark.parametrize("i_dt", [np.int8, np.int16, np.int32,
253                                      np.int64, np.int_])
254    def test_multiply_integer_dtypes(self, i_dt, dt):
255        a = np.array("abc", dtype=dt)
256        i = np.array(3, dtype=i_dt)
257        res = np.array("abcabcabc", dtype=dt)
258        assert_array_equal(np.strings.multiply(a, i), res)
259
260    @pytest.mark.parametrize("in_,out", [
261        ("", False),
262        ("a", True),
263        ("A", True),
264        ("\n", False),
265        ("abc", True),
266        ("aBc123", False),
267        ("abc\n", False),
268        (["abc", "aBc123"], [True, False]),
269    ])
270    def test_isalpha(self, in_, out, dt):
271        in_ = np.array(in_, dtype=dt)
272        assert_array_equal(np.strings.isalpha(in_), out)
273
274    @pytest.mark.parametrize("in_,out", [
275        ('', False),
276        ('a', True),
277        ('A', True),
278        ('\n', False),
279        ('123abc456', True),
280        ('a1b3c', True),
281        ('aBc000 ', False),
282        ('abc\n', False),
283    ])
284    def test_isalnum(self, in_, out, dt):
285        in_ = np.array(in_, dtype=dt)
286        assert_array_equal(np.strings.isalnum(in_), out)
287
288    @pytest.mark.parametrize("in_,out", [
289        ("", False),
290        ("a", False),
291        ("0", True),
292        ("012345", True),
293        ("012345a", False),
294        (["a", "012345"], [False, True]),
295    ])
296    def test_isdigit(self, in_, out, dt):
297        in_ = np.array(in_, dtype=dt)
298        assert_array_equal(np.strings.isdigit(in_), out)
299
300    @pytest.mark.parametrize("in_,out", [
301        ("", False),
302        ("a", False),
303        ("1", False),
304        (" ", True),
305        ("\t", True),
306        ("\r", True),
307        ("\n", True),
308        (" \t\r \n", True),
309        (" \t\r\na", False),
310        (["\t1", " \t\r \n"], [False, True])
311    ])
312    def test_isspace(self, in_, out, dt):
313        in_ = np.array(in_, dtype=dt)
314        assert_array_equal(np.strings.isspace(in_), out)
315
316    @pytest.mark.parametrize("in_,out", [
317        ('', False),
318        ('a', True),
319        ('A', False),
320        ('\n', False),
321        ('abc', True),
322        ('aBc', False),
323        ('abc\n', True),
324    ])
325    def test_islower(self, in_, out, dt):
326        in_ = np.array(in_, dtype=dt)
327        assert_array_equal(np.strings.islower(in_), out)
328
329    @pytest.mark.parametrize("in_,out", [
330        ('', False),
331        ('a', False),
332        ('A', True),
333        ('\n', False),
334        ('ABC', True),
335        ('AbC', False),
336        ('ABC\n', True),
337    ])
338    def test_isupper(self, in_, out, dt):
339        in_ = np.array(in_, dtype=dt)
340        assert_array_equal(np.strings.isupper(in_), out)
341
342    @pytest.mark.parametrize("in_,out", [
343        ('', False),
344        ('a', False),
345        ('A', True),
346        ('\n', False),
347        ('A Titlecased Line', True),
348        ('A\nTitlecased Line', True),
349        ('A Titlecased, Line', True),
350        ('Not a capitalized String', False),
351        ('Not\ta Titlecase String', False),
352        ('Not--a Titlecase String', False),
353        ('NOT', False),
354    ])
355    def test_istitle(self, in_, out, dt):
356        in_ = np.array(in_, dtype=dt)
357        assert_array_equal(np.strings.istitle(in_), out)
358
359    @pytest.mark.parametrize("in_,out", [
360        ("", 0),
361        ("abc", 3),
362        ("12345", 5),
363        ("MixedCase", 9),
364        ("12345 \x00 ", 8),
365        ("UPPER", 5),
366        (["abc", "12345 \x00 "], [3, 8]),
367    ])
368    def test_str_len(self, in_, out, dt):
369        in_ = np.array(in_, dtype=dt)
370        assert_array_equal(np.strings.str_len(in_), out)
371
372    @pytest.mark.parametrize("a,sub,start,end,out", [
373        ("abcdefghiabc", "abc", 0, None, 0),
374        ("abcdefghiabc", "abc", 1, None, 9),
375        ("abcdefghiabc", "def", 4, None, -1),
376        ("abc", "", 0, None, 0),
377        ("abc", "", 3, None, 3),
378        ("abc", "", 4, None, -1),
379        ("rrarrrrrrrrra", "a", 0, None, 2),
380        ("rrarrrrrrrrra", "a", 4, None, 12),
381        ("rrarrrrrrrrra", "a", 4, 6, -1),
382        ("", "", 0, None, 0),
383        ("", "", 1, 1, -1),
384        ("", "", MAX, 0, -1),
385        ("", "xx", 0, None, -1),
386        ("", "xx", 1, 1, -1),
387        ("", "xx", MAX, 0, -1),
388        pytest.param(99 * "a" + "b", "b", 0, None, 99,
389                     id="99*a+b-b-0-None-99"),
390        pytest.param(98 * "a" + "ba", "ba", 0, None, 98,
391                     id="98*a+ba-ba-0-None-98"),
392        pytest.param(100 * "a", "b", 0, None, -1,
393                     id="100*a-b-0-None--1"),
394        pytest.param(30000 * "a" + 100 * "b", 100 * "b", 0, None, 30000,
395                     id="30000*a+100*b-100*b-0-None-30000"),
396        pytest.param(30000 * "a", 100 * "b", 0, None, -1,
397                     id="30000*a-100*b-0-None--1"),
398        pytest.param(15000 * "a" + 15000 * "b", 15000 * "b", 0, None, 15000,
399                     id="15000*a+15000*b-15000*b-0-None-15000"),
400        pytest.param(15000 * "a" + 15000 * "b", 15000 * "c", 0, None, -1,
401                     id="15000*a+15000*b-15000*c-0-None--1"),
402        (["abcdefghiabc", "rrarrrrrrrrra"], ["def", "arr"], [0, 3],
403         None, [3, -1]),
404        ("Ae¢☃€ 😊" * 2, "😊", 0, None, 6),
405        ("Ae¢☃€ 😊" * 2, "😊", 7, None, 13),
406        pytest.param("A" * (2 ** 17), r"[\w]+\Z", 0, None, -1,
407                     id=r"A*2**17-[\w]+\Z-0-None--1"),
408    ])
409    def test_find(self, a, sub, start, end, out, dt):
410        if "😊" in a and dt == "S":
411            pytest.skip("Bytes dtype does not support non-ascii input")
412        a = np.array(a, dtype=dt)
413        sub = np.array(sub, dtype=dt)
414        assert_array_equal(np.strings.find(a, sub, start, end), out)
415
416    @pytest.mark.parametrize("a,sub,start,end,out", [
417        ("abcdefghiabc", "abc", 0, None, 9),
418        ("abcdefghiabc", "", 0, None, 12),
419        ("abcdefghiabc", "abcd", 0, None, 0),
420        ("abcdefghiabc", "abcz", 0, None, -1),
421        ("abc", "", 0, None, 3),
422        ("abc", "", 3, None, 3),
423        ("abc", "", 4, None, -1),
424        ("rrarrrrrrrrra", "a", 0, None, 12),
425        ("rrarrrrrrrrra", "a", 4, None, 12),
426        ("rrarrrrrrrrra", "a", 4, 6, -1),
427        (["abcdefghiabc", "rrarrrrrrrrra"], ["abc", "a"], [0, 0],
428         None, [9, 12]),
429        ("Ae¢☃€ 😊" * 2, "😊", 0, None, 13),
430        ("Ae¢☃€ 😊" * 2, "😊", 0, 7, 6),
431    ])
432    def test_rfind(self, a, sub, start, end, out, dt):
433        if "😊" in a and dt == "S":
434            pytest.skip("Bytes dtype does not support non-ascii input")
435        a = np.array(a, dtype=dt)
436        sub = np.array(sub, dtype=dt)
437        assert_array_equal(np.strings.rfind(a, sub, start, end), out)
438
439    @pytest.mark.parametrize("a,sub,start,end,out", [
440        ("aaa", "a", 0, None, 3),
441        ("aaa", "b", 0, None, 0),
442        ("aaa", "a", 1, None, 2),
443        ("aaa", "a", 10, None, 0),
444        ("aaa", "a", -1, None, 1),
445        ("aaa", "a", -10, None, 3),
446        ("aaa", "a", 0, 1, 1),
447        ("aaa", "a", 0, 10, 3),
448        ("aaa", "a", 0, -1, 2),
449        ("aaa", "a", 0, -10, 0),
450        ("aaa", "", 1, None, 3),
451        ("aaa", "", 3, None, 1),
452        ("aaa", "", 10, None, 0),
453        ("aaa", "", -1, None, 2),
454        ("aaa", "", -10, None, 4),
455        ("aaa", "aaaa", 0, None, 0),
456        pytest.param(98 * "a" + "ba", "ba", 0, None, 1,
457                     id="98*a+ba-ba-0-None-1"),
458        pytest.param(30000 * "a" + 100 * "b", 100 * "b", 0, None, 1,
459                     id="30000*a+100*b-100*b-0-None-1"),
460        pytest.param(30000 * "a", 100 * "b", 0, None, 0,
461                     id="30000*a-100*b-0-None-0"),
462        pytest.param(30000 * "a" + 100 * "ab", "ab", 0, None, 100,
463                     id="30000*a+100*ab-ab-0-None-100"),
464        pytest.param(15000 * "a" + 15000 * "b", 15000 * "b", 0, None, 1,
465                     id="15000*a+15000*b-15000*b-0-None-1"),
466        pytest.param(15000 * "a" + 15000 * "b", 15000 * "c", 0, None, 0,
467                     id="15000*a+15000*b-15000*c-0-None-0"),
468        ("", "", 0, None, 1),
469        ("", "", 1, 1, 0),
470        ("", "", MAX, 0, 0),
471        ("", "xx", 0, None, 0),
472        ("", "xx", 1, 1, 0),
473        ("", "xx", MAX, 0, 0),
474        (["aaa", ""], ["a", ""], [0, 0], None, [3, 1]),
475        ("Ae¢☃€ 😊" * 100, "😊", 0, None, 100),
476    ])
477    def test_count(self, a, sub, start, end, out, dt):
478        if "😊" in a and dt == "S":
479            pytest.skip("Bytes dtype does not support non-ascii input")
480        a = np.array(a, dtype=dt)
481        sub = np.array(sub, dtype=dt)
482        assert_array_equal(np.strings.count(a, sub, start, end), out)
483
484    @pytest.mark.parametrize("a,prefix,start,end,out", [
485        ("hello", "he", 0, None, True),
486        ("hello", "hello", 0, None, True),
487        ("hello", "hello world", 0, None, False),
488        ("hello", "", 0, None, True),
489        ("hello", "ello", 0, None, False),
490        ("hello", "ello", 1, None, True),
491        ("hello", "o", 4, None, True),
492        ("hello", "o", 5, None, False),
493        ("hello", "", 5, None, True),
494        ("hello", "lo", 6, None, False),
495        ("helloworld", "lowo", 3, None, True),
496        ("helloworld", "lowo", 3, 7, True),
497        ("helloworld", "lowo", 3, 6, False),
498        ("", "", 0, 1, True),
499        ("", "", 0, 0, True),
500        ("", "", 1, 0, False),
501        ("hello", "he", 0, -1, True),
502        ("hello", "he", -53, -1, True),
503        ("hello", "hello", 0, -1, False),
504        ("hello", "hello world", -1, -10, False),
505        ("hello", "ello", -5, None, False),
506        ("hello", "ello", -4, None, True),
507        ("hello", "o", -2, None, False),
508        ("hello", "o", -1, None, True),
509        ("hello", "", -3, -3, True),
510        ("hello", "lo", -9, None, False),
511        (["hello", ""], ["he", ""], [0, 0], None, [True, True]),
512    ])
513    def test_startswith(self, a, prefix, start, end, out, dt):
514        a = np.array(a, dtype=dt)
515        prefix = np.array(prefix, dtype=dt)
516        assert_array_equal(np.strings.startswith(a, prefix, start, end), out)
517
518    @pytest.mark.parametrize("a,suffix,start,end,out", [
519        ("hello", "lo", 0, None, True),
520        ("hello", "he", 0, None, False),
521        ("hello", "", 0, None, True),
522        ("hello", "hello world", 0, None, False),
523        ("helloworld", "worl", 0, None, False),
524        ("helloworld", "worl", 3, 9, True),
525        ("helloworld", "world", 3, 12, True),
526        ("helloworld", "lowo", 1, 7, True),
527        ("helloworld", "lowo", 2, 7, True),
528        ("helloworld", "lowo", 3, 7, True),
529        ("helloworld", "lowo", 4, 7, False),
530        ("helloworld", "lowo", 3, 8, False),
531        ("ab", "ab", 0, 1, False),
532        ("ab", "ab", 0, 0, False),
533        ("", "", 0, 1, True),
534        ("", "", 0, 0, True),
535        ("", "", 1, 0, False),
536        ("hello", "lo", -2, None, True),
537        ("hello", "he", -2, None, False),
538        ("hello", "", -3, -3, True),
539        ("hello", "hello world", -10, -2, False),
540        ("helloworld", "worl", -6, None, False),
541        ("helloworld", "worl", -5, -1, True),
542        ("helloworld", "worl", -5, 9, True),
543        ("helloworld", "world", -7, 12, True),
544        ("helloworld", "lowo", -99, -3, True),
545        ("helloworld", "lowo", -8, -3, True),
546        ("helloworld", "lowo", -7, -3, True),
547        ("helloworld", "lowo", 3, -4, False),
548        ("helloworld", "lowo", -8, -2, False),
549        (["hello", "helloworld"], ["lo", "worl"], [0, -6], None,
550         [True, False]),
551    ])
552    def test_endswith(self, a, suffix, start, end, out, dt):
553        a = np.array(a, dtype=dt)
554        suffix = np.array(suffix, dtype=dt)
555        assert_array_equal(np.strings.endswith(a, suffix, start, end), out)
556
557    @pytest.mark.parametrize("a,chars,out", [
558        ("", None, ""),
559        ("   hello   ", None, "hello   "),
560        ("hello", None, "hello"),
561        (" \t\n\r\f\vabc \t\n\r\f\v", None, "abc \t\n\r\f\v"),
562        (["   hello   ", "hello"], None, ["hello   ", "hello"]),
563        ("", "", ""),
564        ("", "xyz", ""),
565        ("hello", "", "hello"),
566        ("xyzzyhelloxyzzy", "xyz", "helloxyzzy"),
567        ("hello", "xyz", "hello"),
568        ("xyxz", "xyxz", ""),
569        ("xyxzx", "x", "yxzx"),
570        (["xyzzyhelloxyzzy", "hello"], ["xyz", "xyz"],
571         ["helloxyzzy", "hello"]),
572        (["ba", "ac", "baa", "bba"], "b", ["a", "ac", "aa", "a"]),
573    ])
574    def test_lstrip(self, a, chars, out, dt):
575        a = np.array(a, dtype=dt)
576        out = np.array(out, dtype=dt)
577        if chars is not None:
578            chars = np.array(chars, dtype=dt)
579            assert_array_equal(np.strings.lstrip(a, chars), out)
580        else:
581            assert_array_equal(np.strings.lstrip(a), out)
582
583    @pytest.mark.parametrize("a,chars,out", [
584        ("", None, ""),
585        ("   hello   ", None, "   hello"),
586        ("hello", None, "hello"),
587        (" \t\n\r\f\vabc \t\n\r\f\v", None, " \t\n\r\f\vabc"),
588        (["   hello   ", "hello"], None, ["   hello", "hello"]),
589        ("", "", ""),
590        ("", "xyz", ""),
591        ("hello", "", "hello"),
592        (["hello    ", "abcdefghijklmnop"], None,
593         ["hello", "abcdefghijklmnop"]),
594        ("xyzzyhelloxyzzy", "xyz", "xyzzyhello"),
595        ("hello", "xyz", "hello"),
596        ("xyxz", "xyxz", ""),
597        ("    ", None, ""),
598        ("xyxzx", "x", "xyxz"),
599        (["xyzzyhelloxyzzy", "hello"], ["xyz", "xyz"],
600         ["xyzzyhello", "hello"]),
601        (["ab", "ac", "aab", "abb"], "b", ["a", "ac", "aa", "a"]),
602    ])
603    def test_rstrip(self, a, chars, out, dt):
604        a = np.array(a, dtype=dt)
605        out = np.array(out, dtype=dt)
606        if chars is not None:
607            chars = np.array(chars, dtype=dt)
608            assert_array_equal(np.strings.rstrip(a, chars), out)
609        else:
610            assert_array_equal(np.strings.rstrip(a), out)
611
612    @pytest.mark.parametrize("a,chars,out", [
613        ("", None, ""),
614        ("   hello   ", None, "hello"),
615        ("hello", None, "hello"),
616        (" \t\n\r\f\vabc \t\n\r\f\v", None, "abc"),
617        (["   hello   ", "hello"], None, ["hello", "hello"]),
618        ("", "", ""),
619        ("", "xyz", ""),
620        ("hello", "", "hello"),
621        ("xyzzyhelloxyzzy", "xyz", "hello"),
622        ("hello", "xyz", "hello"),
623        ("xyxz", "xyxz", ""),
624        ("xyxzx", "x", "yxz"),
625        (["xyzzyhelloxyzzy", "hello"], ["xyz", "xyz"],
626         ["hello", "hello"]),
627        (["bab", "ac", "baab", "bbabb"], "b", ["a", "ac", "aa", "a"]),
628    ])
629    def test_strip(self, a, chars, out, dt):
630        a = np.array(a, dtype=dt)
631        if chars is not None:
632            chars = np.array(chars, dtype=dt)
633        out = np.array(out, dtype=dt)
634        assert_array_equal(np.strings.strip(a, chars), out)
635
636    @pytest.mark.parametrize("buf,old,new,count,res", [
637        ("", "", "", -1, ""),
638        ("", "", "A", -1, "A"),
639        ("", "A", "", -1, ""),
640        ("", "A", "A", -1, ""),
641        ("", "", "", 100, ""),
642        ("", "", "A", 100, "A"),
643        ("A", "", "", -1, "A"),
644        ("A", "", "*", -1, "*A*"),
645        ("A", "", "*1", -1, "*1A*1"),
646        ("A", "", "*-#", -1, "*-#A*-#"),
647        ("AA", "", "*-", -1, "*-A*-A*-"),
648        ("AA", "", "*-", -1, "*-A*-A*-"),
649        ("AA", "", "*-", 4, "*-A*-A*-"),
650        ("AA", "", "*-", 3, "*-A*-A*-"),
651        ("AA", "", "*-", 2, "*-A*-A"),
652        ("AA", "", "*-", 1, "*-AA"),
653        ("AA", "", "*-", 0, "AA"),
654        ("A", "A", "", -1, ""),
655        ("AAA", "A", "", -1, ""),
656        ("AAA", "A", "", -1, ""),
657        ("AAA", "A", "", 4, ""),
658        ("AAA", "A", "", 3, ""),
659        ("AAA", "A", "", 2, "A"),
660        ("AAA", "A", "", 1, "AA"),
661        ("AAA", "A", "", 0, "AAA"),
662        ("AAAAAAAAAA", "A", "", -1, ""),
663        ("ABACADA", "A", "", -1, "BCD"),
664        ("ABACADA", "A", "", -1, "BCD"),
665        ("ABACADA", "A", "", 5, "BCD"),
666        ("ABACADA", "A", "", 4, "BCD"),
667        ("ABACADA", "A", "", 3, "BCDA"),
668        ("ABACADA", "A", "", 2, "BCADA"),
669        ("ABACADA", "A", "", 1, "BACADA"),
670        ("ABACADA", "A", "", 0, "ABACADA"),
671        ("ABCAD", "A", "", -1, "BCD"),
672        ("ABCADAA", "A", "", -1, "BCD"),
673        ("BCD", "A", "", -1, "BCD"),
674        ("*************", "A", "", -1, "*************"),
675        ("^" + "A" * 1000 + "^", "A", "", 999, "^A^"),
676        ("the", "the", "", -1, ""),
677        ("theater", "the", "", -1, "ater"),
678        ("thethe", "the", "", -1, ""),
679        ("thethethethe", "the", "", -1, ""),
680        ("theatheatheathea", "the", "", -1, "aaaa"),
681        ("that", "the", "", -1, "that"),
682        ("thaet", "the", "", -1, "thaet"),
683        ("here and there", "the", "", -1, "here and re"),
684        ("here and there and there", "the", "", -1, "here and re and re"),
685        ("here and there and there", "the", "", 3, "here and re and re"),
686        ("here and there and there", "the", "", 2, "here and re and re"),
687        ("here and there and there", "the", "", 1, "here and re and there"),
688        ("here and there and there", "the", "", 0, "here and there and there"),
689        ("here and there and there", "the", "", -1, "here and re and re"),
690        ("abc", "the", "", -1, "abc"),
691        ("abcdefg", "the", "", -1, "abcdefg"),
692        ("bbobob", "bob", "", -1, "bob"),
693        ("bbobobXbbobob", "bob", "", -1, "bobXbob"),
694        ("aaaaaaabob", "bob", "", -1, "aaaaaaa"),
695        ("aaaaaaa", "bob", "", -1, "aaaaaaa"),
696        ("Who goes there?", "o", "o", -1, "Who goes there?"),
697        ("Who goes there?", "o", "O", -1, "WhO gOes there?"),
698        ("Who goes there?", "o", "O", -1, "WhO gOes there?"),
699        ("Who goes there?", "o", "O", 3, "WhO gOes there?"),
700        ("Who goes there?", "o", "O", 2, "WhO gOes there?"),
701        ("Who goes there?", "o", "O", 1, "WhO goes there?"),
702        ("Who goes there?", "o", "O", 0, "Who goes there?"),
703        ("Who goes there?", "a", "q", -1, "Who goes there?"),
704        ("Who goes there?", "W", "w", -1, "who goes there?"),
705        ("WWho goes there?WW", "W", "w", -1, "wwho goes there?ww"),
706        ("Who goes there?", "?", "!", -1, "Who goes there!"),
707        ("Who goes there??", "?", "!", -1, "Who goes there!!"),
708        ("Who goes there?", ".", "!", -1, "Who goes there?"),
709        ("This is a tissue", "is", "**", -1, "Th** ** a t**sue"),
710        ("This is a tissue", "is", "**", -1, "Th** ** a t**sue"),
711        ("This is a tissue", "is", "**", 4, "Th** ** a t**sue"),
712        ("This is a tissue", "is", "**", 3, "Th** ** a t**sue"),
713        ("This is a tissue", "is", "**", 2, "Th** ** a tissue"),
714        ("This is a tissue", "is", "**", 1, "Th** is a tissue"),
715        ("This is a tissue", "is", "**", 0, "This is a tissue"),
716        ("bobob", "bob", "cob", -1, "cobob"),
717        ("bobobXbobobob", "bob", "cob", -1, "cobobXcobocob"),
718        ("bobob", "bot", "bot", -1, "bobob"),
719        ("Reykjavik", "k", "KK", -1, "ReyKKjaviKK"),
720        ("Reykjavik", "k", "KK", -1, "ReyKKjaviKK"),
721        ("Reykjavik", "k", "KK", 2, "ReyKKjaviKK"),
722        ("Reykjavik", "k", "KK", 1, "ReyKKjavik"),
723        ("Reykjavik", "k", "KK", 0, "Reykjavik"),
724        ("A.B.C.", ".", "----", -1, "A----B----C----"),
725        ("Reykjavik", "q", "KK", -1, "Reykjavik"),
726        ("spam, spam, eggs and spam", "spam", "ham", -1,
727            "ham, ham, eggs and ham"),
728        ("spam, spam, eggs and spam", "spam", "ham", -1,
729            "ham, ham, eggs and ham"),
730        ("spam, spam, eggs and spam", "spam", "ham", 4,
731            "ham, ham, eggs and ham"),
732        ("spam, spam, eggs and spam", "spam", "ham", 3,
733            "ham, ham, eggs and ham"),
734        ("spam, spam, eggs and spam", "spam", "ham", 2,
735            "ham, ham, eggs and spam"),
736        ("spam, spam, eggs and spam", "spam", "ham", 1,
737            "ham, spam, eggs and spam"),
738        ("spam, spam, eggs and spam", "spam", "ham", 0,
739            "spam, spam, eggs and spam"),
740        ("bobobob", "bobob", "bob", -1, "bobob"),
741        ("bobobobXbobobob", "bobob", "bob", -1, "bobobXbobob"),
742        ("BOBOBOB", "bob", "bobby", -1, "BOBOBOB"),
743        ("one!two!three!", "!", "@", 1, "one@two!three!"),
744        ("one!two!three!", "!", "", -1, "onetwothree"),
745        ("one!two!three!", "!", "@", 2, "one@two@three!"),
746        ("one!two!three!", "!", "@", 3, "one@two@three@"),
747        ("one!two!three!", "!", "@", 4, "one@two@three@"),
748        ("one!two!three!", "!", "@", 0, "one!two!three!"),
749        ("one!two!three!", "!", "@", -1, "one@two@three@"),
750        ("one!two!three!", "x", "@", -1, "one!two!three!"),
751        ("one!two!three!", "x", "@", 2, "one!two!three!"),
752        ("abc", "", "-", -1, "-a-b-c-"),
753        ("abc", "", "-", 3, "-a-b-c"),
754        ("abc", "", "-", 0, "abc"),
755        ("abc", "ab", "--", 0, "abc"),
756        ("abc", "xy", "--", -1, "abc"),
757        (["abbc", "abbd"], "b", "z", [1, 2], ["azbc", "azzd"]),
758    ])
759    def test_replace(self, buf, old, new, count, res, dt):
760        if "😊" in buf and dt == "S":
761            pytest.skip("Bytes dtype does not support non-ascii input")
762        buf = np.array(buf, dtype=dt)
763        old = np.array(old, dtype=dt)
764        new = np.array(new, dtype=dt)
765        res = np.array(res, dtype=dt)
766        assert_array_equal(np.strings.replace(buf, old, new, count), res)
767
768    @pytest.mark.parametrize("buf,sub,start,end,res", [
769        ("abcdefghiabc", "", 0, None, 0),
770        ("abcdefghiabc", "def", 0, None, 3),
771        ("abcdefghiabc", "abc", 0, None, 0),
772        ("abcdefghiabc", "abc", 1, None, 9),
773    ])
774    def test_index(self, buf, sub, start, end, res, dt):
775        buf = np.array(buf, dtype=dt)
776        sub = np.array(sub, dtype=dt)
777        assert_array_equal(np.strings.index(buf, sub, start, end), res)
778
779    @pytest.mark.parametrize("buf,sub,start,end", [
780        ("abcdefghiabc", "hib", 0, None),
781        ("abcdefghiab", "abc", 1, None),
782        ("abcdefghi", "ghi", 8, None),
783        ("abcdefghi", "ghi", -1, None),
784        ("rrarrrrrrrrra", "a", 4, 6),
785    ])
786    def test_index_raises(self, buf, sub, start, end, dt):
787        buf = np.array(buf, dtype=dt)
788        sub = np.array(sub, dtype=dt)
789        with pytest.raises(ValueError, match="substring not found"):
790            np.strings.index(buf, sub, start, end)
791
792    @pytest.mark.parametrize("buf,sub,start,end,res", [
793        ("abcdefghiabc", "", 0, None, 12),
794        ("abcdefghiabc", "def", 0, None, 3),
795        ("abcdefghiabc", "abc", 0, None, 9),
796        ("abcdefghiabc", "abc", 0, -1, 0),
797    ])
798    def test_rindex(self, buf, sub, start, end, res, dt):
799        buf = np.array(buf, dtype=dt)
800        sub = np.array(sub, dtype=dt)
801        assert_array_equal(np.strings.rindex(buf, sub, start, end), res)
802
803    @pytest.mark.parametrize("buf,sub,start,end", [
804        ("abcdefghiabc", "hib", 0, None),
805        ("defghiabc", "def", 1, None),
806        ("defghiabc", "abc", 0, -1),
807        ("abcdefghi", "ghi", 0, 8),
808        ("abcdefghi", "ghi", 0, -1),
809        ("rrarrrrrrrrra", "a", 4, 6),
810    ])
811    def test_rindex_raises(self, buf, sub, start, end, dt):
812        buf = np.array(buf, dtype=dt)
813        sub = np.array(sub, dtype=dt)
814        with pytest.raises(ValueError, match="substring not found"):
815            np.strings.rindex(buf, sub, start, end)
816
817    @pytest.mark.parametrize("buf,tabsize,res", [
818        ("abc\rab\tdef\ng\thi", 8, "abc\rab      def\ng       hi"),
819        ("abc\rab\tdef\ng\thi", 4, "abc\rab  def\ng   hi"),
820        ("abc\r\nab\tdef\ng\thi", 8, "abc\r\nab      def\ng       hi"),
821        ("abc\r\nab\tdef\ng\thi", 4, "abc\r\nab  def\ng   hi"),
822        ("abc\r\nab\r\ndef\ng\r\nhi", 4, "abc\r\nab\r\ndef\ng\r\nhi"),
823        (" \ta\n\tb", 1, "  a\n b"),
824    ])
825    def test_expandtabs(self, buf, tabsize, res, dt):
826        buf = np.array(buf, dtype=dt)
827        res = np.array(res, dtype=dt)
828        assert_array_equal(np.strings.expandtabs(buf, tabsize), res)
829
830    def test_expandtabs_raises_overflow(self, dt):
831        with pytest.raises(OverflowError, match="new string is too long"):
832            np.strings.expandtabs(np.array("\ta\n\tb", dtype=dt), sys.maxsize)
833            np.strings.expandtabs(np.array("\ta\n\tb", dtype=dt), 2**61)
834
835    def test_expandtabs_length_not_cause_segfault(self, dt):
836        # see gh-28829
837        with pytest.raises(
838            _UFuncNoLoopError,
839            match="did not contain a loop with signature matching types",
840        ):
841            np._core.strings._expandtabs_length.reduce(np.zeros(200))
842
843        with pytest.raises(
844            _UFuncNoLoopError,
845            match="did not contain a loop with signature matching types",
846        ):
847            np.strings.expandtabs(np.zeros(200))
848
849    FILL_ERROR = "The fill character must be exactly one character long"
850
851    def test_center_raises_multiple_character_fill(self, dt):
852        buf = np.array("abc", dtype=dt)
853        fill = np.array("**", dtype=dt)
854        with pytest.raises(TypeError, match=self.FILL_ERROR):
855            np.strings.center(buf, 10, fill)
856
857    def test_ljust_raises_multiple_character_fill(self, dt):
858        buf = np.array("abc", dtype=dt)
859        fill = np.array("**", dtype=dt)
860        with pytest.raises(TypeError, match=self.FILL_ERROR):
861            np.strings.ljust(buf, 10, fill)
862
863    def test_rjust_raises_multiple_character_fill(self, dt):
864        buf = np.array("abc", dtype=dt)
865        fill = np.array("**", dtype=dt)
866        with pytest.raises(TypeError, match=self.FILL_ERROR):
867            np.strings.rjust(buf, 10, fill)
868
869    @pytest.mark.parametrize("buf,width,fillchar,res", [
870        ('abc', 10, ' ', '   abc    '),
871        ('abc', 6, ' ', ' abc  '),
872        ('abc', 3, ' ', 'abc'),
873        ('abc', 2, ' ', 'abc'),
874        ('abc', -2, ' ', 'abc'),
875        ('abc', 10, '*', '***abc****'),
876    ])
877    def test_center(self, buf, width, fillchar, res, dt):
878        buf = np.array(buf, dtype=dt)
879        fillchar = np.array(fillchar, dtype=dt)
880        res = np.array(res, dtype=dt)
881        assert_array_equal(np.strings.center(buf, width, fillchar), res)
882
883    @pytest.mark.parametrize("buf,width,fillchar,res", [
884        ('abc', 10, ' ', 'abc       '),
885        ('abc', 6, ' ', 'abc   '),
886        ('abc', 3, ' ', 'abc'),
887        ('abc', 2, ' ', 'abc'),
888        ('abc', -2, ' ', 'abc'),
889        ('abc', 10, '*', 'abc*******'),
890    ])
891    def test_ljust(self, buf, width, fillchar, res, dt):
892        buf = np.array(buf, dtype=dt)
893        fillchar = np.array(fillchar, dtype=dt)
894        res = np.array(res, dtype=dt)
895        assert_array_equal(np.strings.ljust(buf, width, fillchar), res)
896
897    @pytest.mark.parametrize("buf,width,fillchar,res", [
898        ('abc', 10, ' ', '       abc'),
899        ('abc', 6, ' ', '   abc'),
900        ('abc', 3, ' ', 'abc'),
901        ('abc', 2, ' ', 'abc'),
902        ('abc', -2, ' ', 'abc'),
903        ('abc', 10, '*', '*******abc'),
904    ])
905    def test_rjust(self, buf, width, fillchar, res, dt):
906        buf = np.array(buf, dtype=dt)
907        fillchar = np.array(fillchar, dtype=dt)
908        res = np.array(res, dtype=dt)
909        assert_array_equal(np.strings.rjust(buf, width, fillchar), res)
910
911    @pytest.mark.parametrize("buf,width,res", [
912        ('123', 2, '123'),
913        ('123', 3, '123'),
914        ('0123', 4, '0123'),
915        ('+123', 3, '+123'),
916        ('+123', 4, '+123'),
917        ('+123', 5, '+0123'),
918        ('+0123', 5, '+0123'),
919        ('-123', 3, '-123'),
920        ('-123', 4, '-123'),
921        ('-0123', 5, '-0123'),
922        ('000', 3, '000'),
923        ('34', 1, '34'),
924        ('34', -1, '34'),
925        ('0034', 4, '0034'),
926    ])
927    def test_zfill(self, buf, width, res, dt):
928        buf = np.array(buf, dtype=dt)
929        res = np.array(res, dtype=dt)
930        assert_array_equal(np.strings.zfill(buf, width), res)
931
932    @pytest.mark.parametrize("buf,sep,res1,res2,res3", [
933        ("this is the partition method", "ti", "this is the par",
934            "ti", "tion method"),
935        ("http://www.python.org", "://", "http", "://", "www.python.org"),
936        ("http://www.python.org", "?", "http://www.python.org", "", ""),
937        ("http://www.python.org", "http://", "", "http://", "www.python.org"),
938        ("http://www.python.org", "org", "http://www.python.", "org", ""),
939        ("http://www.python.org", ["://", "?", "http://", "org"],
940            ["http", "http://www.python.org", "", "http://www.python."],
941            ["://", "", "http://", "org"],
942            ["www.python.org", "", "www.python.org", ""]),
943        ("mississippi", "ss", "mi", "ss", "issippi"),
944        ("mississippi", "i", "m", "i", "ssissippi"),
945        ("mississippi", "w", "mississippi", "", ""),
946    ])
947    def test_partition(self, buf, sep, res1, res2, res3, dt):
948        buf = np.array(buf, dtype=dt)
949        sep = np.array(sep, dtype=dt)
950        res1 = np.array(res1, dtype=dt)
951        res2 = np.array(res2, dtype=dt)
952        res3 = np.array(res3, dtype=dt)
953        act1, act2, act3 = np.strings.partition(buf, sep)
954        assert_array_equal(act1, res1)
955        assert_array_equal(act2, res2)
956        assert_array_equal(act3, res3)
957        assert_array_equal(act1 + act2 + act3, buf)
958
959    @pytest.mark.parametrize("buf,sep,res1,res2,res3", [
960        ("this is the partition method", "ti", "this is the parti",
961            "ti", "on method"),
962        ("http://www.python.org", "://", "http", "://", "www.python.org"),
963        ("http://www.python.org", "?", "", "", "http://www.python.org"),
964        ("http://www.python.org", "http://", "", "http://", "www.python.org"),
965        ("http://www.python.org", "org", "http://www.python.", "org", ""),
966        ("http://www.python.org", ["://", "?", "http://", "org"],
967            ["http", "", "", "http://www.python."],
968            ["://", "", "http://", "org"],
969            ["www.python.org", "http://www.python.org", "www.python.org", ""]),
970        ("mississippi", "ss", "missi", "ss", "ippi"),
971        ("mississippi", "i", "mississipp", "i", ""),
972        ("mississippi", "w", "", "", "mississippi"),
973    ])
974    def test_rpartition(self, buf, sep, res1, res2, res3, dt):
975        buf = np.array(buf, dtype=dt)
976        sep = np.array(sep, dtype=dt)
977        res1 = np.array(res1, dtype=dt)
978        res2 = np.array(res2, dtype=dt)
979        res3 = np.array(res3, dtype=dt)
980        act1, act2, act3 = np.strings.rpartition(buf, sep)
981        assert_array_equal(act1, res1)
982        assert_array_equal(act2, res2)
983        assert_array_equal(act3, res3)
984        assert_array_equal(act1 + act2 + act3, buf)
985
986    @pytest.mark.parametrize("args", [
987        (None,),
988        (None, None),
989        (None, None, -1),
990        (0,),
991        (0, None),
992        (0, None, -1),
993        (1,),
994        (1, None),
995        (1, None, -1),
996        (3,),
997        (3, None),
998        (5,),
999        (5, None),
1000        (5, 5),
1001        (5, 5, -1),
1002        (6,),  # test index past the end
1003        (6, None),
1004        (6, None, -1),
1005        (6, 7),  # test start and stop index past the end
1006        (4, 3),  # test start > stop index
1007        (-1,),
1008        (-1, None),
1009        (-1, None, -1),
1010        (-3,),
1011        (-3, None),
1012        ([3, 4],),
1013        ([3, 4], None),
1014        ([2, 4],),
1015        ([-3, 5],),
1016        ([-3, 5], None),
1017        ([-3, 5], None, -1),
1018        ([0, -5],),
1019        ([0, -5], None),
1020        ([0, -5], None, -1),
1021        (1, 4),
1022        (-3, 5),
1023        (None, -1),
1024        (0, [4, 2]),
1025        ([1, 2], [-1, -2]),
1026        (1, 5, 2),
1027        (None, None, -1),
1028        ([0, 6], [-1, 0], [2, -1]),
1029    ])
1030    @pytest.mark.parametrize("buf", [
1031        ["hello", "world"],
1032        ['hello world', 'γεια σου κόσμε', '你好世界', '👋 🌍'],
1033    ])
1034    def test_slice(self, args, buf, dt):
1035        if dt == "S" and "你好世界" in buf:
1036            pytest.skip("Bytes dtype does not support non-ascii input")
1037        if len(buf) == 4:
1038            args = tuple(s * 2 if isinstance(s, list) else s for s in args)
1039        buf = np.array(buf, dtype=dt)
1040        act = np.strings.slice(buf, *args)
1041        bcast_args = tuple(np.broadcast_to(arg, buf.shape) for arg in args)
1042        res = np.array([s[slice(*arg)]
1043                        for s, arg in zip(buf, zip(*bcast_args))],
1044                       dtype=dt)
1045        assert_array_equal(act, res)
1046
1047    def test_slice_unsupported(self, dt):
1048        with pytest.raises(TypeError, match="did not contain a loop"):
1049            np.strings.slice(np.array([1, 2, 3]), 4)
1050
1051        regexp = (r"Cannot cast ufunc '_slice' input .* "
1052                  r"from .* to dtype\('int(64|32)'\)")
1053        with pytest.raises(TypeError, match=regexp):
1054            np.strings.slice(np.array(['foo', 'bar'], dtype=dt),
1055                             np.array(['foo', 'bar'], dtype=dt))
1056
1057    @pytest.mark.parametrize("int_dt", [np.int8, np.int16, np.int32,
1058                                        np.int64, np.uint8, np.uint16,
1059                                        np.uint32, np.uint64])
1060    def test_slice_int_type_promotion(self, int_dt, dt):
1061        buf = np.array(["hello", "world"], dtype=dt)
1062        np_slice = np.strings.slice
1063        assert_array_equal(np_slice(buf, int_dt(4)),
1064                           np.array(["hell", "worl"], dtype=dt))
1065        assert_array_equal(np_slice(buf, np.array([4, 4], dtype=int_dt)),
1066                           np.array(["hell", "worl"], dtype=dt))
1067
1068        assert_array_equal(np_slice(buf, int_dt(2), int_dt(4)),
1069                           np.array(["ll", "rl"], dtype=dt))
1070        assert_array_equal(np_slice(buf, np.array([2, 2], dtype=int_dt),
1071                                    np.array([4, 4], dtype=int_dt)),
1072                           np.array(["ll", "rl"], dtype=dt))
1073
1074        assert_array_equal(np_slice(buf, int_dt(0), int_dt(4), int_dt(2)),
1075                           np.array(["hl", "wr"], dtype=dt))
1076        assert_array_equal(np_slice(buf,
1077                                    np.array([0, 0], dtype=int_dt),
1078                                    np.array([4, 4], dtype=int_dt),
1079                                    np.array([2, 2], dtype=int_dt)),
1080                           np.array(["hl", "wr"], dtype=dt))
1081
1082@pytest.mark.parametrize("dt", ["U", "T"])
1083class TestMethodsWithUnicode:
1084    @pytest.mark.parametrize("in_,out", [
1085        ("", False),
1086        ("a", False),
1087        ("0", True),
1088        ("\u2460", False),  # CIRCLED DIGIT 1
1089        ("\xbc", False),  # VULGAR FRACTION ONE QUARTER
1090        ("\u0660", True),  # ARABIC_INDIC DIGIT ZERO
1091        ("012345", True),
1092        ("012345a", False),
1093        (["0", "a"], [True, False]),
1094    ])
1095    def test_isdecimal_unicode(self, in_, out, dt):
1096        buf = np.array(in_, dtype=dt)
1097        assert_array_equal(np.strings.isdecimal(buf), out)
1098
1099    @pytest.mark.parametrize("in_,out", [
1100        ("", False),
1101        ("a", False),
1102        ("0", True),
1103        ("\u2460", True),  # CIRCLED DIGIT 1
1104        ("\xbc", True),  # VULGAR FRACTION ONE QUARTER
1105        ("\u0660", True),  # ARABIC_INDIC DIGIT ZERO
1106        ("012345", True),
1107        ("012345a", False),
1108        (["0", "a"], [True, False]),
1109    ])
1110    def test_isnumeric_unicode(self, in_, out, dt):
1111        buf = np.array(in_, dtype=dt)
1112        assert_array_equal(np.strings.isnumeric(buf), out)
1113
1114    @pytest.mark.parametrize("buf,old,new,count,res", [
1115        ("...\u043c......<", "<", "&lt;", -1, "...\u043c......&lt;"),
1116        ("Ae¢☃€ 😊" * 2, "A", "B", -1, "Be¢☃€ 😊Be¢☃€ 😊"),
1117        ("Ae¢☃€ 😊" * 2, "😊", "B", -1, "Ae¢☃€ BAe¢☃€ B"),
1118    ])
1119    def test_replace_unicode(self, buf, old, new, count, res, dt):
1120        buf = np.array(buf, dtype=dt)
1121        old = np.array(old, dtype=dt)
1122        new = np.array(new, dtype=dt)
1123        res = np.array(res, dtype=dt)
1124        assert_array_equal(np.strings.replace(buf, old, new, count), res)
1125
1126    @pytest.mark.parametrize("in_", [
1127        '\U00010401',
1128        '\U00010427',
1129        '\U00010429',
1130        '\U0001044E',
1131        '\U0001D7F6',
1132        '\U00011066',
1133        '\U000104A0',
1134        pytest.param('\U0001F107', marks=pytest.mark.xfail(
1135            sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1136            reason="PYPY bug in Py_UNICODE_ISALNUM",
1137            strict=True)),
1138    ])
1139    def test_isalnum_unicode(self, in_, dt):
1140        in_ = np.array(in_, dtype=dt)
1141        assert_array_equal(np.strings.isalnum(in_), True)
1142
1143    @pytest.mark.parametrize("in_,out", [
1144        ('\u1FFc', False),
1145        ('\u2167', False),
1146        ('\U00010401', False),
1147        ('\U00010427', False),
1148        ('\U0001F40D', False),
1149        ('\U0001F46F', False),
1150        ('\u2177', True),
1151        pytest.param('\U00010429', True, marks=pytest.mark.xfail(
1152            sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1153            reason="PYPY bug in Py_UNICODE_ISLOWER",
1154            strict=True)),
1155        ('\U0001044E', True),
1156    ])
1157    def test_islower_unicode(self, in_, out, dt):
1158        in_ = np.array(in_, dtype=dt)
1159        assert_array_equal(np.strings.islower(in_), out)
1160
1161    @pytest.mark.parametrize("in_,out", [
1162        ('\u1FFc', False),
1163        ('\u2167', True),
1164        ('\U00010401', True),
1165        ('\U00010427', True),
1166        ('\U0001F40D', False),
1167        ('\U0001F46F', False),
1168        ('\u2177', False),
1169        pytest.param('\U00010429', False, marks=pytest.mark.xfail(
1170            sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1171            reason="PYPY bug in Py_UNICODE_ISUPPER",
1172            strict=True)),
1173        ('\U0001044E', False),
1174    ])
1175    def test_isupper_unicode(self, in_, out, dt):
1176        in_ = np.array(in_, dtype=dt)
1177        assert_array_equal(np.strings.isupper(in_), out)
1178
1179    @pytest.mark.parametrize("in_,out", [
1180        ('\u1FFc', True),
1181        ('Greek \u1FFcitlecases ...', True),
1182        pytest.param('\U00010401\U00010429', True, marks=pytest.mark.xfail(
1183            sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1184            reason="PYPY bug in Py_UNICODE_ISISTITLE",
1185            strict=True)),
1186        ('\U00010427\U0001044E', True),
1187        pytest.param('\U00010429', False, marks=pytest.mark.xfail(
1188            sys.platform == 'win32' and IS_PYPY_LT_7_3_16,
1189            reason="PYPY bug in Py_UNICODE_ISISTITLE",
1190            strict=True)),
1191        ('\U0001044E', False),
1192        ('\U0001F40D', False),
1193        ('\U0001F46F', False),
1194    ])
1195    def test_istitle_unicode(self, in_, out, dt):
1196        in_ = np.array(in_, dtype=dt)
1197        assert_array_equal(np.strings.istitle(in_), out)
1198
1199    @pytest.mark.parametrize("buf,sub,start,end,res", [
1200        ("Ae¢☃€ 😊" * 2, "😊", 0, None, 6),

Showing the first 1,200 of 1524 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai