codekingpro/portable-devtools
115k
1"""
2This module contains a set of functions for vectorized string
3operations.
4"""
5
6import functools
7import sys
8
9import numpy as np
10from numpy import (
11 add,
12 equal,
13 greater,
14 greater_equal,
15 less,
16 less_equal,
17 multiply as _multiply_ufunc,
18 not_equal,
19)
20from numpy._core.multiarray import _vec_string
21from numpy._core.overrides import array_function_dispatch, set_module
22from numpy._core.umath import (
23 _center,
24 _expandtabs,
25 _expandtabs_length,
26 _ljust,
27 _lstrip_chars,
28 _lstrip_whitespace,
29 _partition,
30 _partition_index,
31 _replace,
32 _rjust,
33 _rpartition,
34 _rpartition_index,
35 _rstrip_chars,
36 _rstrip_whitespace,
37 _slice,
38 _strip_chars,
39 _strip_whitespace,
40 _zfill,
41 count as _count_ufunc,
42 endswith as _endswith_ufunc,
43 find as _find_ufunc,
44 index as _index_ufunc,
45 isalnum,
46 isalpha,
47 isdecimal,
48 isdigit,
49 islower,
50 isnumeric,
51 isspace,
52 istitle,
53 isupper,
54 rfind as _rfind_ufunc,
55 rindex as _rindex_ufunc,
56 startswith as _startswith_ufunc,
57 str_len,
58)
59
60
61def _override___module__():
62 for ufunc in [
63 isalnum, isalpha, isdecimal, isdigit, islower, isnumeric, isspace,
64 istitle, isupper, str_len,
65 ]:
66 ufunc.__module__ = "numpy.strings"
67 ufunc.__qualname__ = ufunc.__name__
68
69
70_override___module__()
71
72
73__all__ = [
74 # UFuncs
75 "equal", "not_equal", "less", "less_equal", "greater", "greater_equal",
76 "add", "multiply", "isalpha", "isdigit", "isspace", "isalnum", "islower",
77 "isupper", "istitle", "isdecimal", "isnumeric", "str_len", "find",
78 "rfind", "index", "rindex", "count", "startswith", "endswith", "lstrip",
79 "rstrip", "strip", "replace", "expandtabs", "center", "ljust", "rjust",
80 "zfill", "partition", "rpartition", "slice",
81
82 # _vec_string - Will gradually become ufuncs as well
83 "upper", "lower", "swapcase", "capitalize", "title",
84
85 # _vec_string - Will probably not become ufuncs
86 "mod", "decode", "encode", "translate",
87
88 # Removed from namespace until behavior has been crystallized
89 # "join", "split", "rsplit", "splitlines",
90]
91
92
93MAX = np.iinfo(np.int64).max
94
95array_function_dispatch = functools.partial(
96 array_function_dispatch, module='numpy.strings')
97
98
99def _get_num_chars(a):
100 """
101 Helper function that returns the number of characters per field in
102 a string or unicode array. This is to abstract out the fact that
103 for a unicode array this is itemsize / 4.
104 """
105 if issubclass(a.dtype.type, np.str_):
106 return a.itemsize // 4
107 return a.itemsize
108
109
110def _to_bytes_or_str_array(result, output_dtype_like):
111 """
112 Helper function to cast a result back into an array
113 with the appropriate dtype if an object array must be used
114 as an intermediary.
115 """
116 output_dtype_like = np.asarray(output_dtype_like)
117 if result.size == 0:
118 # Calling asarray & tolist in an empty array would result
119 # in losing shape information
120 return result.astype(output_dtype_like.dtype)
121 ret = np.asarray(result.tolist())
122 if isinstance(output_dtype_like.dtype, np.dtypes.StringDType):
123 return ret.astype(type(output_dtype_like.dtype))
124 return ret.astype(type(output_dtype_like.dtype)(_get_num_chars(ret)))
125
126
127def _clean_args(*args):
128 """
129 Helper function for delegating arguments to Python string
130 functions.
131
132 Many of the Python string operations that have optional arguments
133 do not use 'None' to indicate a default value. In these cases,
134 we need to remove all None arguments, and those following them.
135 """
136 newargs = []
137 for chk in args:
138 if chk is None:
139 break
140 newargs.append(chk)
141 return newargs
142
143
144def _multiply_dispatcher(a, i):
145 return (a,)
146
147
148@set_module("numpy.strings")
149@array_function_dispatch(_multiply_dispatcher)
150def multiply(a, i):
151 """
152 Return (a * i), that is string multiple concatenation,
153 element-wise.
154
155 Values in ``i`` of less than 0 are treated as 0 (which yields an
156 empty string).
157
158 Parameters
159 ----------
160 a : array_like, with ``StringDType``, ``bytes_`` or ``str_`` dtype
161
162 i : array_like, with any integer dtype
163
164 Returns
165 -------
166 out : ndarray
167 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
168 depending on input types
169
170 Examples
171 --------
172 >>> import numpy as np
173 >>> a = np.array(["a", "b", "c"])
174 >>> np.strings.multiply(a, 3)
175 array(['aaa', 'bbb', 'ccc'], dtype='<U3')
176 >>> i = np.array([1, 2, 3])
177 >>> np.strings.multiply(a, i)
178 array(['a', 'bb', 'ccc'], dtype='<U3')
179 >>> np.strings.multiply(np.array(['a']), i)
180 array(['a', 'aa', 'aaa'], dtype='<U3')
181 >>> a = np.array(['a', 'b', 'c', 'd', 'e', 'f']).reshape((2, 3))
182 >>> np.strings.multiply(a, 3)
183 array([['aaa', 'bbb', 'ccc'],
184 ['ddd', 'eee', 'fff']], dtype='<U3')
185 >>> np.strings.multiply(a, i)
186 array([['a', 'bb', 'ccc'],
187 ['d', 'ee', 'fff']], dtype='<U3')
188
189 """
190 a = np.asanyarray(a)
191
192 i = np.asanyarray(i)
193 if not np.issubdtype(i.dtype, np.integer):
194 raise TypeError(f"unsupported type {i.dtype} for operand 'i'")
195 i = np.maximum(i, 0)
196
197 # delegate to stringdtype loops that also do overflow checking
198 if a.dtype.char == "T":
199 return a * i
200
201 a_len = str_len(a)
202
203 # Ensure we can do a_len * i without overflow.
204 if np.any(a_len > sys.maxsize / np.maximum(i, 1)):
205 raise OverflowError("Overflow encountered in string multiply")
206
207 buffersizes = a_len * i
208 out_dtype = f"{a.dtype.char}{buffersizes.max()}"
209 out = np.empty_like(a, shape=buffersizes.shape, dtype=out_dtype)
210 return _multiply_ufunc(a, i, out=out)
211
212
213def _mod_dispatcher(a, values):
214 return (a, values)
215
216
217@set_module("numpy.strings")
218@array_function_dispatch(_mod_dispatcher)
219def mod(a, values):
220 """
221 Return (a % i), that is pre-Python 2.6 string formatting
222 (interpolation), element-wise for a pair of array_likes of str
223 or unicode.
224
225 Parameters
226 ----------
227 a : array_like, with `np.bytes_` or `np.str_` dtype
228
229 values : array_like of values
230 These values will be element-wise interpolated into the string.
231
232 Returns
233 -------
234 out : ndarray
235 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
236 depending on input types
237
238 Examples
239 --------
240 >>> import numpy as np
241 >>> a = np.array(["NumPy is a %s library"])
242 >>> np.strings.mod(a, values=["Python"])
243 array(['NumPy is a Python library'], dtype='<U25')
244
245 >>> a = np.array([b'%d bytes', b'%d bits'])
246 >>> values = np.array([8, 64])
247 >>> np.strings.mod(a, values)
248 array([b'8 bytes', b'64 bits'], dtype='|S7')
249
250 """
251 return _to_bytes_or_str_array(
252 _vec_string(a, np.object_, '__mod__', (values,)), a)
253
254
255@set_module("numpy.strings")
256def find(a, sub, start=0, end=None):
257 """
258 For each element, return the lowest index in the string where
259 substring ``sub`` is found, such that ``sub`` is contained in the
260 range [``start``, ``end``).
261
262 Parameters
263 ----------
264 a : array_like, with ``StringDType``, ``bytes_`` or ``str_`` dtype
265
266 sub : array_like, with `np.bytes_` or `np.str_` dtype
267 The substring to search for.
268
269 start, end : array_like, with any integer dtype
270 The range to look in, interpreted as in slice notation.
271
272 Returns
273 -------
274 y : ndarray
275 Output array of ints
276
277 See Also
278 --------
279 str.find
280
281 Examples
282 --------
283 >>> import numpy as np
284 >>> a = np.array(["NumPy is a Python library"])
285 >>> np.strings.find(a, "Python")
286 array([11])
287
288 """
289 end = end if end is not None else MAX
290 return _find_ufunc(a, sub, start, end)
291
292
293@set_module("numpy.strings")
294def rfind(a, sub, start=0, end=None):
295 """
296 For each element, return the highest index in the string where
297 substring ``sub`` is found, such that ``sub`` is contained in the
298 range [``start``, ``end``).
299
300 Parameters
301 ----------
302 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
303
304 sub : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
305 The substring to search for.
306
307 start, end : array_like, with any integer dtype
308 The range to look in, interpreted as in slice notation.
309
310 Returns
311 -------
312 y : ndarray
313 Output array of ints
314
315 See Also
316 --------
317 str.rfind
318
319 Examples
320 --------
321 >>> import numpy as np
322 >>> a = np.array(["Computer Science"])
323 >>> np.strings.rfind(a, "Science", start=0, end=None)
324 array([9])
325 >>> np.strings.rfind(a, "Science", start=0, end=8)
326 array([-1])
327 >>> b = np.array(["Computer Science", "Science"])
328 >>> np.strings.rfind(b, "Science", start=0, end=None)
329 array([9, 0])
330
331 """
332 end = end if end is not None else MAX
333 return _rfind_ufunc(a, sub, start, end)
334
335
336@set_module("numpy.strings")
337def index(a, sub, start=0, end=None):
338 """
339 Like `find`, but raises :exc:`ValueError` when the substring is not found.
340
341 Parameters
342 ----------
343 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
344
345 sub : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
346
347 start, end : array_like, with any integer dtype, optional
348
349 Returns
350 -------
351 out : ndarray
352 Output array of ints.
353
354 See Also
355 --------
356 find, str.index
357
358 Examples
359 --------
360 >>> import numpy as np
361 >>> a = np.array(["Computer Science"])
362 >>> np.strings.index(a, "Science", start=0, end=None)
363 array([9])
364
365 """
366 end = end if end is not None else MAX
367 return _index_ufunc(a, sub, start, end)
368
369
370@set_module("numpy.strings")
371def rindex(a, sub, start=0, end=None):
372 """
373 Like `rfind`, but raises :exc:`ValueError` when the substring `sub` is
374 not found.
375
376 Parameters
377 ----------
378 a : array-like, with `np.bytes_` or `np.str_` dtype
379
380 sub : array-like, with `np.bytes_` or `np.str_` dtype
381
382 start, end : array-like, with any integer dtype, optional
383
384 Returns
385 -------
386 out : ndarray
387 Output array of ints.
388
389 See Also
390 --------
391 rfind, str.rindex
392
393 Examples
394 --------
395 >>> a = np.array(["Computer Science"])
396 >>> np.strings.rindex(a, "Science", start=0, end=None)
397 array([9])
398
399 """
400 end = end if end is not None else MAX
401 return _rindex_ufunc(a, sub, start, end)
402
403
404@set_module("numpy.strings")
405def count(a, sub, start=0, end=None):
406 """
407 Returns an array with the number of non-overlapping occurrences of
408 substring ``sub`` in the range [``start``, ``end``).
409
410 Parameters
411 ----------
412 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
413
414 sub : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
415 The substring to search for.
416
417 start, end : array_like, with any integer dtype
418 The range to look in, interpreted as in slice notation.
419
420 Returns
421 -------
422 y : ndarray
423 Output array of ints
424
425 See Also
426 --------
427 str.count
428
429 Examples
430 --------
431 >>> import numpy as np
432 >>> c = np.array(['aAaAaA', ' aA ', 'abBABba'])
433 >>> c
434 array(['aAaAaA', ' aA ', 'abBABba'], dtype='<U7')
435 >>> np.strings.count(c, 'A')
436 array([3, 1, 1])
437 >>> np.strings.count(c, 'aA')
438 array([3, 1, 0])
439 >>> np.strings.count(c, 'A', start=1, end=4)
440 array([2, 1, 1])
441 >>> np.strings.count(c, 'A', start=1, end=3)
442 array([1, 0, 0])
443
444 """
445 end = end if end is not None else MAX
446 return _count_ufunc(a, sub, start, end)
447
448
449@set_module("numpy.strings")
450def startswith(a, prefix, start=0, end=None):
451 """
452 Returns a boolean array which is `True` where the string element
453 in ``a`` starts with ``prefix``, otherwise `False`.
454
455 Parameters
456 ----------
457 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
458
459 prefix : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
460
461 start, end : array_like, with any integer dtype
462 With ``start``, test beginning at that position. With ``end``,
463 stop comparing at that position.
464
465 Returns
466 -------
467 out : ndarray
468 Output array of bools
469
470 See Also
471 --------
472 str.startswith
473
474 Examples
475 --------
476 >>> import numpy as np
477 >>> s = np.array(['foo', 'bar'])
478 >>> s
479 array(['foo', 'bar'], dtype='<U3')
480 >>> np.strings.startswith(s, 'fo')
481 array([True, False])
482 >>> np.strings.startswith(s, 'o', start=1, end=2)
483 array([True, False])
484
485 """
486 end = end if end is not None else MAX
487 return _startswith_ufunc(a, prefix, start, end)
488
489
490@set_module("numpy.strings")
491def endswith(a, suffix, start=0, end=None):
492 """
493 Returns a boolean array which is `True` where the string element
494 in ``a`` ends with ``suffix``, otherwise `False`.
495
496 Parameters
497 ----------
498 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
499
500 suffix : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
501
502 start, end : array_like, with any integer dtype
503 With ``start``, test beginning at that position. With ``end``,
504 stop comparing at that position.
505
506 Returns
507 -------
508 out : ndarray
509 Output array of bools
510
511 See Also
512 --------
513 str.endswith
514
515 Examples
516 --------
517 >>> import numpy as np
518 >>> s = np.array(['foo', 'bar'])
519 >>> s
520 array(['foo', 'bar'], dtype='<U3')
521 >>> np.strings.endswith(s, 'ar')
522 array([False, True])
523 >>> np.strings.endswith(s, 'a', start=1, end=2)
524 array([False, True])
525
526 """
527 end = end if end is not None else MAX
528 return _endswith_ufunc(a, suffix, start, end)
529
530
531def _code_dispatcher(a, encoding=None, errors=None):
532 return (a,)
533
534
535@set_module("numpy.strings")
536@array_function_dispatch(_code_dispatcher)
537def decode(a, encoding=None, errors=None):
538 r"""
539 Calls :meth:`bytes.decode` element-wise.
540
541 The set of available codecs comes from the Python standard library,
542 and may be extended at runtime. For more information, see the
543 :mod:`codecs` module.
544
545 Parameters
546 ----------
547 a : array_like, with ``bytes_`` dtype
548
549 encoding : str, optional
550 The name of an encoding
551
552 errors : str, optional
553 Specifies how to handle encoding errors
554
555 Returns
556 -------
557 out : ndarray
558
559 See Also
560 --------
561 :py:meth:`bytes.decode`
562
563 Notes
564 -----
565 The type of the result will depend on the encoding specified.
566
567 Examples
568 --------
569 >>> import numpy as np
570 >>> c = np.array([b'\x81\xc1\x81\xc1\x81\xc1', b'@@\x81\xc1@@',
571 ... b'\x81\x82\xc2\xc1\xc2\x82\x81'])
572 >>> c
573 array([b'\x81\xc1\x81\xc1\x81\xc1', b'@@\x81\xc1@@',
574 b'\x81\x82\xc2\xc1\xc2\x82\x81'], dtype='|S7')
575 >>> np.strings.decode(c, encoding='cp037')
576 array(['aAaAaA', ' aA ', 'abBABba'], dtype='<U7')
577
578 """
579 return _to_bytes_or_str_array(
580 _vec_string(a, np.object_, 'decode', _clean_args(encoding, errors)),
581 np.str_(''))
582
583
584@set_module("numpy.strings")
585@array_function_dispatch(_code_dispatcher)
586def encode(a, encoding=None, errors=None):
587 """
588 Calls :meth:`str.encode` element-wise.
589
590 The set of available codecs comes from the Python standard library,
591 and may be extended at runtime. For more information, see the
592 :mod:`codecs` module.
593
594 Parameters
595 ----------
596 a : array_like, with ``StringDType`` or ``str_`` dtype
597
598 encoding : str, optional
599 The name of an encoding
600
601 errors : str, optional
602 Specifies how to handle encoding errors
603
604 Returns
605 -------
606 out : ndarray
607
608 See Also
609 --------
610 str.encode
611
612 Notes
613 -----
614 The type of the result will depend on the encoding specified.
615
616 Examples
617 --------
618 >>> import numpy as np
619 >>> a = np.array(['aAaAaA', ' aA ', 'abBABba'])
620 >>> np.strings.encode(a, encoding='cp037')
621 array([b'\x81\xc1\x81\xc1\x81\xc1', b'@@\x81\xc1@@',
622 b'\x81\x82\xc2\xc1\xc2\x82\x81'], dtype='|S7')
623
624 """
625 return _to_bytes_or_str_array(
626 _vec_string(a, np.object_, 'encode', _clean_args(encoding, errors)),
627 np.bytes_(b''))
628
629
630def _expandtabs_dispatcher(a, tabsize=None):
631 return (a,)
632
633
634@set_module("numpy.strings")
635@array_function_dispatch(_expandtabs_dispatcher)
636def expandtabs(a, tabsize=8):
637 """
638 Return a copy of each string element where all tab characters are
639 replaced by one or more spaces.
640
641 Calls :meth:`str.expandtabs` element-wise.
642
643 Return a copy of each string element where all tab characters are
644 replaced by one or more spaces, depending on the current column
645 and the given `tabsize`. The column number is reset to zero after
646 each newline occurring in the string. This doesn't understand other
647 non-printing characters or escape sequences.
648
649 Parameters
650 ----------
651 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
652 Input array
653 tabsize : int, optional
654 Replace tabs with `tabsize` number of spaces. If not given defaults
655 to 8 spaces.
656
657 Returns
658 -------
659 out : ndarray
660 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
661 depending on input type
662
663 See Also
664 --------
665 str.expandtabs
666
667 Examples
668 --------
669 >>> import numpy as np
670 >>> a = np.array(['\t\tHello\tworld'])
671 >>> np.strings.expandtabs(a, tabsize=4) # doctest: +SKIP
672 array([' Hello world'], dtype='<U21') # doctest: +SKIP
673
674 """
675 a = np.asanyarray(a)
676 tabsize = np.asanyarray(tabsize)
677
678 if a.dtype.char == "T":
679 return _expandtabs(a, tabsize)
680
681 buffersizes = _expandtabs_length(a, tabsize)
682 out_dtype = f"{a.dtype.char}{buffersizes.max()}"
683 out = np.empty_like(a, shape=buffersizes.shape, dtype=out_dtype)
684 return _expandtabs(a, tabsize, out=out)
685
686
687def _just_dispatcher(a, width, fillchar=None):
688 return (a,)
689
690
691@set_module("numpy.strings")
692@array_function_dispatch(_just_dispatcher)
693def center(a, width, fillchar=' '):
694 """
695 Return a copy of `a` with its elements centered in a string of
696 length `width`.
697
698 Parameters
699 ----------
700 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
701
702 width : array_like, with any integer dtype
703 The length of the resulting strings, unless ``width < str_len(a)``.
704 fillchar : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
705 Optional padding character to use (default is space).
706
707 Returns
708 -------
709 out : ndarray
710 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
711 depending on input types
712
713 See Also
714 --------
715 str.center
716
717 Notes
718 -----
719 While it is possible for ``a`` and ``fillchar`` to have different dtypes,
720 passing a non-ASCII character in ``fillchar`` when ``a`` is of dtype "S"
721 is not allowed, and a ``ValueError`` is raised.
722
723 Examples
724 --------
725 >>> import numpy as np
726 >>> c = np.array(['a1b2','1b2a','b2a1','2a1b']); c
727 array(['a1b2', '1b2a', 'b2a1', '2a1b'], dtype='<U4')
728 >>> np.strings.center(c, width=9)
729 array([' a1b2 ', ' 1b2a ', ' b2a1 ', ' 2a1b '], dtype='<U9')
730 >>> np.strings.center(c, width=9, fillchar='*')
731 array(['***a1b2**', '***1b2a**', '***b2a1**', '***2a1b**'], dtype='<U9')
732 >>> np.strings.center(c, width=1)
733 array(['a1b2', '1b2a', 'b2a1', '2a1b'], dtype='<U4')
734
735 """
736 width = np.asanyarray(width)
737
738 if not np.issubdtype(width.dtype, np.integer):
739 raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
740
741 a = np.asanyarray(a)
742 fillchar = np.asanyarray(fillchar)
743
744 if np.any(str_len(fillchar) != 1):
745 raise TypeError(
746 "The fill character must be exactly one character long")
747
748 if np.result_type(a, fillchar).char == "T":
749 return _center(a, width, fillchar)
750
751 fillchar = fillchar.astype(a.dtype, copy=False)
752 width = np.maximum(str_len(a), width)
753 out_dtype = f"{a.dtype.char}{width.max()}"
754 shape = np.broadcast_shapes(a.shape, width.shape, fillchar.shape)
755 out = np.empty_like(a, shape=shape, dtype=out_dtype)
756
757 return _center(a, width, fillchar, out=out)
758
759
760@set_module("numpy.strings")
761@array_function_dispatch(_just_dispatcher)
762def ljust(a, width, fillchar=' '):
763 """
764 Return an array with the elements of `a` left-justified in a
765 string of length `width`.
766
767 Parameters
768 ----------
769 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
770
771 width : array_like, with any integer dtype
772 The length of the resulting strings, unless ``width < str_len(a)``.
773 fillchar : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
774 Optional character to use for padding (default is space).
775
776 Returns
777 -------
778 out : ndarray
779 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
780 depending on input types
781
782 See Also
783 --------
784 str.ljust
785
786 Notes
787 -----
788 While it is possible for ``a`` and ``fillchar`` to have different dtypes,
789 passing a non-ASCII character in ``fillchar`` when ``a`` is of dtype "S"
790 is not allowed, and a ``ValueError`` is raised.
791
792 Examples
793 --------
794 >>> import numpy as np
795 >>> c = np.array(['aAaAaA', ' aA ', 'abBABba'])
796 >>> np.strings.ljust(c, width=3)
797 array(['aAaAaA', ' aA ', 'abBABba'], dtype='<U7')
798 >>> np.strings.ljust(c, width=9)
799 array(['aAaAaA ', ' aA ', 'abBABba '], dtype='<U9')
800
801 """
802 width = np.asanyarray(width)
803 if not np.issubdtype(width.dtype, np.integer):
804 raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
805
806 a = np.asanyarray(a)
807 fillchar = np.asanyarray(fillchar)
808
809 if np.any(str_len(fillchar) != 1):
810 raise TypeError(
811 "The fill character must be exactly one character long")
812
813 if np.result_type(a, fillchar).char == "T":
814 return _ljust(a, width, fillchar)
815
816 fillchar = fillchar.astype(a.dtype, copy=False)
817 width = np.maximum(str_len(a), width)
818 shape = np.broadcast_shapes(a.shape, width.shape, fillchar.shape)
819 out_dtype = f"{a.dtype.char}{width.max()}"
820 out = np.empty_like(a, shape=shape, dtype=out_dtype)
821
822 return _ljust(a, width, fillchar, out=out)
823
824
825@set_module("numpy.strings")
826@array_function_dispatch(_just_dispatcher)
827def rjust(a, width, fillchar=' '):
828 """
829 Return an array with the elements of `a` right-justified in a
830 string of length `width`.
831
832 Parameters
833 ----------
834 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
835
836 width : array_like, with any integer dtype
837 The length of the resulting strings, unless ``width < str_len(a)``.
838 fillchar : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
839 Optional padding character to use (default is space).
840
841 Returns
842 -------
843 out : ndarray
844 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
845 depending on input types
846
847 See Also
848 --------
849 str.rjust
850
851 Notes
852 -----
853 While it is possible for ``a`` and ``fillchar`` to have different dtypes,
854 passing a non-ASCII character in ``fillchar`` when ``a`` is of dtype "S"
855 is not allowed, and a ``ValueError`` is raised.
856
857 Examples
858 --------
859 >>> import numpy as np
860 >>> a = np.array(['aAaAaA', ' aA ', 'abBABba'])
861 >>> np.strings.rjust(a, width=3)
862 array(['aAaAaA', ' aA ', 'abBABba'], dtype='<U7')
863 >>> np.strings.rjust(a, width=9)
864 array([' aAaAaA', ' aA ', ' abBABba'], dtype='<U9')
865
866 """
867 width = np.asanyarray(width)
868 if not np.issubdtype(width.dtype, np.integer):
869 raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
870
871 a = np.asanyarray(a)
872 fillchar = np.asanyarray(fillchar)
873
874 if np.any(str_len(fillchar) != 1):
875 raise TypeError(
876 "The fill character must be exactly one character long")
877
878 if np.result_type(a, fillchar).char == "T":
879 return _rjust(a, width, fillchar)
880
881 fillchar = fillchar.astype(a.dtype, copy=False)
882 width = np.maximum(str_len(a), width)
883 shape = np.broadcast_shapes(a.shape, width.shape, fillchar.shape)
884 out_dtype = f"{a.dtype.char}{width.max()}"
885 out = np.empty_like(a, shape=shape, dtype=out_dtype)
886
887 return _rjust(a, width, fillchar, out=out)
888
889
890def _zfill_dispatcher(a, width):
891 return (a,)
892
893
894@set_module("numpy.strings")
895@array_function_dispatch(_zfill_dispatcher)
896def zfill(a, width):
897 """
898 Return the numeric string left-filled with zeros. A leading
899 sign prefix (``+``/``-``) is handled by inserting the padding
900 after the sign character rather than before.
901
902 Parameters
903 ----------
904 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
905
906 width : array_like, with any integer dtype
907 Width of string to left-fill elements in `a`.
908
909 Returns
910 -------
911 out : ndarray
912 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
913 depending on input type
914
915 See Also
916 --------
917 str.zfill
918
919 Examples
920 --------
921 >>> import numpy as np
922 >>> np.strings.zfill(['1', '-1', '+1'], 3)
923 array(['001', '-01', '+01'], dtype='<U3')
924
925 """
926 width = np.asanyarray(width)
927 if not np.issubdtype(width.dtype, np.integer):
928 raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
929
930 a = np.asanyarray(a)
931
932 if a.dtype.char == "T":
933 return _zfill(a, width)
934
935 width = np.maximum(str_len(a), width)
936 shape = np.broadcast_shapes(a.shape, width.shape)
937 out_dtype = f"{a.dtype.char}{width.max()}"
938 out = np.empty_like(a, shape=shape, dtype=out_dtype)
939 return _zfill(a, width, out=out)
940
941
942@set_module("numpy.strings")
943def lstrip(a, chars=None):
944 """
945 For each element in `a`, return a copy with the leading characters
946 removed.
947
948 Parameters
949 ----------
950 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
951 chars : scalar with the same dtype as ``a``, optional
952 The ``chars`` argument is a string specifying the set of
953 characters to be removed. If ``None``, the ``chars``
954 argument defaults to removing whitespace. The ``chars`` argument
955 is not a prefix or suffix; rather, all combinations of its
956 values are stripped.
957
958 Returns
959 -------
960 out : ndarray
961 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
962 depending on input types
963
964 See Also
965 --------
966 str.lstrip
967
968 Examples
969 --------
970 >>> import numpy as np
971 >>> c = np.array(['aAaAaA', ' aA ', 'abBABba'])
972 >>> c
973 array(['aAaAaA', ' aA ', 'abBABba'], dtype='<U7')
974 # The 'a' variable is unstripped from c[1] because of leading whitespace.
975 >>> np.strings.lstrip(c, 'a')
976 array(['AaAaA', ' aA ', 'bBABba'], dtype='<U7')
977 >>> np.strings.lstrip(c, 'A') # leaves c unchanged
978 array(['aAaAaA', ' aA ', 'abBABba'], dtype='<U7')
979 >>> (np.strings.lstrip(c, ' ') == np.strings.lstrip(c, '')).all()
980 np.False_
981 >>> (np.strings.lstrip(c, ' ') == np.strings.lstrip(c)).all()
982 np.True_
983
984 """
985 if chars is None:
986 return _lstrip_whitespace(a)
987 return _lstrip_chars(a, chars)
988
989
990@set_module("numpy.strings")
991def rstrip(a, chars=None):
992 """
993 For each element in `a`, return a copy with the trailing characters
994 removed.
995
996 Parameters
997 ----------
998 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
999 chars : scalar with the same dtype as ``a``, optional
1000 The ``chars`` argument is a string specifying the set of
1001 characters to be removed. If ``None``, the ``chars``
1002 argument defaults to removing whitespace. The ``chars`` argument
1003 is not a prefix or suffix; rather, all combinations of its
1004 values are stripped.
1005
1006 Returns
1007 -------
1008 out : ndarray
1009 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1010 depending on input types
1011
1012 See Also
1013 --------
1014 str.rstrip
1015
1016 Examples
1017 --------
1018 >>> import numpy as np
1019 >>> c = np.array(['aAaAaA', 'abBABba'])
1020 >>> c
1021 array(['aAaAaA', 'abBABba'], dtype='<U7')
1022 >>> np.strings.rstrip(c, 'a')
1023 array(['aAaAaA', 'abBABb'], dtype='<U7')
1024 >>> np.strings.rstrip(c, 'A')
1025 array(['aAaAa', 'abBABba'], dtype='<U7')
1026
1027 """
1028 if chars is None:
1029 return _rstrip_whitespace(a)
1030 return _rstrip_chars(a, chars)
1031
1032
1033@set_module("numpy.strings")
1034def strip(a, chars=None):
1035 """
1036 For each element in `a`, return a copy with the leading and
1037 trailing characters removed.
1038
1039 Parameters
1040 ----------
1041 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1042 chars : scalar with the same dtype as ``a``, optional
1043 The ``chars`` argument is a string specifying the set of
1044 characters to be removed. If ``None``, the ``chars``
1045 argument defaults to removing whitespace. The ``chars`` argument
1046 is not a prefix or suffix; rather, all combinations of its
1047 values are stripped.
1048
1049 Returns
1050 -------
1051 out : ndarray
1052 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1053 depending on input types
1054
1055 See Also
1056 --------
1057 str.strip
1058
1059 Examples
1060 --------
1061 >>> import numpy as np
1062 >>> c = np.array(['aAaAaA', ' aA ', 'abBABba'])
1063 >>> c
1064 array(['aAaAaA', ' aA ', 'abBABba'], dtype='<U7')
1065 >>> np.strings.strip(c)
1066 array(['aAaAaA', 'aA', 'abBABba'], dtype='<U7')
1067 # 'a' unstripped from c[1] because of leading whitespace.
1068 >>> np.strings.strip(c, 'a')
1069 array(['AaAaA', ' aA ', 'bBABb'], dtype='<U7')
1070 # 'A' unstripped from c[1] because of trailing whitespace.
1071 >>> np.strings.strip(c, 'A')
1072 array(['aAaAa', ' aA ', 'abBABba'], dtype='<U7')
1073
1074 """
1075 if chars is None:
1076 return _strip_whitespace(a)
1077 return _strip_chars(a, chars)
1078
1079
1080def _unary_op_dispatcher(a):
1081 return (a,)
1082
1083
1084@set_module("numpy.strings")
1085@array_function_dispatch(_unary_op_dispatcher)
1086def upper(a):
1087 """
1088 Return an array with the elements converted to uppercase.
1089
1090 Calls :meth:`str.upper` element-wise.
1091
1092 For 8-bit strings, this method is locale-dependent.
1093
1094 Parameters
1095 ----------
1096 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1097 Input array.
1098
1099 Returns
1100 -------
1101 out : ndarray
1102 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1103 depending on input types
1104
1105 See Also
1106 --------
1107 str.upper
1108
1109 Examples
1110 --------
1111 >>> import numpy as np
1112 >>> c = np.array(['a1b c', '1bca', 'bca1']); c
1113 array(['a1b c', '1bca', 'bca1'], dtype='<U5')
1114 >>> np.strings.upper(c)
1115 array(['A1B C', '1BCA', 'BCA1'], dtype='<U5')
1116
1117 """
1118 a_arr = np.asarray(a)
1119 return _vec_string(a_arr, a_arr.dtype, 'upper')
1120
1121
1122@set_module("numpy.strings")
1123@array_function_dispatch(_unary_op_dispatcher)
1124def lower(a):
1125 """
1126 Return an array with the elements converted to lowercase.
1127
1128 Call :meth:`str.lower` element-wise.
1129
1130 For 8-bit strings, this method is locale-dependent.
1131
1132 Parameters
1133 ----------
1134 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1135 Input array.
1136
1137 Returns
1138 -------
1139 out : ndarray
1140 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1141 depending on input types
1142
1143 See Also
1144 --------
1145 str.lower
1146
1147 Examples
1148 --------
1149 >>> import numpy as np
1150 >>> c = np.array(['A1B C', '1BCA', 'BCA1']); c
1151 array(['A1B C', '1BCA', 'BCA1'], dtype='<U5')
1152 >>> np.strings.lower(c)
1153 array(['a1b c', '1bca', 'bca1'], dtype='<U5')
1154
1155 """
1156 a_arr = np.asarray(a)
1157 return _vec_string(a_arr, a_arr.dtype, 'lower')
1158
1159
1160@set_module("numpy.strings")
1161@array_function_dispatch(_unary_op_dispatcher)
1162def swapcase(a):
1163 """
1164 Return element-wise a copy of the string with
1165 uppercase characters converted to lowercase and vice versa.
1166
1167 Calls :meth:`str.swapcase` element-wise.
1168
1169 For 8-bit strings, this method is locale-dependent.
1170
1171 Parameters
1172 ----------
1173 a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1174 Input array.
1175
1176 Returns
1177 -------
1178 out : ndarray
1179 Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1180 depending on input types
1181
1182 See Also
1183 --------
1184 str.swapcase
1185
1186 Examples
1187 --------
1188 >>> import numpy as np
1189 >>> c=np.array(['a1B c','1b Ca','b Ca1','cA1b'],'S5'); c
1190 array(['a1B c', '1b Ca', 'b Ca1', 'cA1b'],
1191 dtype='|S5')
1192 >>> np.strings.swapcase(c)
1193 array(['A1b C', '1B cA', 'B cA1', 'Ca1B'],
1194 dtype='|S5')
1195
1196 """
1197 a_arr = np.asarray(a)
1198 return _vec_string(a_arr, a_arr.dtype, 'swapcase')
1199
1200
