Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
strings.py1814 linesDownload Raw Back to _core
1"""
2This module contains a set of functions for vectorized string
3operations.
4"""
5
6import functools
7import sys
8
9import numpy as np
10from numpy import (
11    add,
12    equal,
13    greater,
14    greater_equal,
15    less,
16    less_equal,
17    multiply as _multiply_ufunc,
18    not_equal,
19)
20from numpy._core.multiarray import _vec_string
21from numpy._core.overrides import array_function_dispatch, set_module
22from numpy._core.umath import (
23    _center,
24    _expandtabs,
25    _expandtabs_length,
26    _ljust,
27    _lstrip_chars,
28    _lstrip_whitespace,
29    _partition,
30    _partition_index,
31    _replace,
32    _rjust,
33    _rpartition,
34    _rpartition_index,
35    _rstrip_chars,
36    _rstrip_whitespace,
37    _slice,
38    _strip_chars,
39    _strip_whitespace,
40    _zfill,
41    count as _count_ufunc,
42    endswith as _endswith_ufunc,
43    find as _find_ufunc,
44    index as _index_ufunc,
45    isalnum,
46    isalpha,
47    isdecimal,
48    isdigit,
49    islower,
50    isnumeric,
51    isspace,
52    istitle,
53    isupper,
54    rfind as _rfind_ufunc,
55    rindex as _rindex_ufunc,
56    startswith as _startswith_ufunc,
57    str_len,
58)
59
60
61def _override___module__():
62    for ufunc in [
63        isalnum, isalpha, isdecimal, isdigit, islower, isnumeric, isspace,
64        istitle, isupper, str_len,
65    ]:
66        ufunc.__module__ = "numpy.strings"
67        ufunc.__qualname__ = ufunc.__name__
68
69
70_override___module__()
71
72
73__all__ = [
74    # UFuncs
75    "equal", "not_equal", "less", "less_equal", "greater", "greater_equal",
76    "add", "multiply", "isalpha", "isdigit", "isspace", "isalnum", "islower",
77    "isupper", "istitle", "isdecimal", "isnumeric", "str_len", "find",
78    "rfind", "index", "rindex", "count", "startswith", "endswith", "lstrip",
79    "rstrip", "strip", "replace", "expandtabs", "center", "ljust", "rjust",
80    "zfill", "partition", "rpartition", "slice",
81
82    # _vec_string - Will gradually become ufuncs as well
83    "upper", "lower", "swapcase", "capitalize", "title",
84
85    # _vec_string - Will probably not become ufuncs
86    "mod", "decode", "encode", "translate",
87
88    # Removed from namespace until behavior has been crystallized
89    # "join", "split", "rsplit", "splitlines",
90]
91
92
93MAX = np.iinfo(np.int64).max
94
95array_function_dispatch = functools.partial(
96    array_function_dispatch, module='numpy.strings')
97
98
99def _get_num_chars(a):
100    """
101    Helper function that returns the number of characters per field in
102    a string or unicode array.  This is to abstract out the fact that
103    for a unicode array this is itemsize / 4.
104    """
105    if issubclass(a.dtype.type, np.str_):
106        return a.itemsize // 4
107    return a.itemsize
108
109
110def _to_bytes_or_str_array(result, output_dtype_like):
111    """
112    Helper function to cast a result back into an array
113    with the appropriate dtype if an object array must be used
114    as an intermediary.
115    """
116    output_dtype_like = np.asarray(output_dtype_like)
117    if result.size == 0:
118        # Calling asarray & tolist in an empty array would result
119        # in losing shape information
120        return result.astype(output_dtype_like.dtype)
121    ret = np.asarray(result.tolist())
122    if isinstance(output_dtype_like.dtype, np.dtypes.StringDType):
123        return ret.astype(type(output_dtype_like.dtype))
124    return ret.astype(type(output_dtype_like.dtype)(_get_num_chars(ret)))
125
126
127def _clean_args(*args):
128    """
129    Helper function for delegating arguments to Python string
130    functions.
131
132    Many of the Python string operations that have optional arguments
133    do not use 'None' to indicate a default value.  In these cases,
134    we need to remove all None arguments, and those following them.
135    """
136    newargs = []
137    for chk in args:
138        if chk is None:
139            break
140        newargs.append(chk)
141    return newargs
142
143
144def _multiply_dispatcher(a, i):
145    return (a,)
146
147
148@set_module("numpy.strings")
149@array_function_dispatch(_multiply_dispatcher)
150def multiply(a, i):
151    """
152    Return (a * i), that is string multiple concatenation,
153    element-wise.
154
155    Values in ``i`` of less than 0 are treated as 0 (which yields an
156    empty string).
157
158    Parameters
159    ----------
160    a : array_like, with ``StringDType``, ``bytes_`` or ``str_`` dtype
161
162    i : array_like, with any integer dtype
163
164    Returns
165    -------
166    out : ndarray
167        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
168        depending on input types
169
170    Examples
171    --------
172    >>> import numpy as np
173    >>> a = np.array(["a", "b", "c"])
174    >>> np.strings.multiply(a, 3)
175    array(['aaa', 'bbb', 'ccc'], dtype='<U3')
176    >>> i = np.array([1, 2, 3])
177    >>> np.strings.multiply(a, i)
178    array(['a', 'bb', 'ccc'], dtype='<U3')
179    >>> np.strings.multiply(np.array(['a']), i)
180    array(['a', 'aa', 'aaa'], dtype='<U3')
181    >>> a = np.array(['a', 'b', 'c', 'd', 'e', 'f']).reshape((2, 3))
182    >>> np.strings.multiply(a, 3)
183    array([['aaa', 'bbb', 'ccc'],
184           ['ddd', 'eee', 'fff']], dtype='<U3')
185    >>> np.strings.multiply(a, i)
186    array([['a', 'bb', 'ccc'],
187           ['d', 'ee', 'fff']], dtype='<U3')
188
189    """
190    a = np.asanyarray(a)
191
192    i = np.asanyarray(i)
193    if not np.issubdtype(i.dtype, np.integer):
194        raise TypeError(f"unsupported type {i.dtype} for operand 'i'")
195    i = np.maximum(i, 0)
196
197    # delegate to stringdtype loops that also do overflow checking
198    if a.dtype.char == "T":
199        return a * i
200
201    a_len = str_len(a)
202
203    # Ensure we can do a_len * i without overflow.
204    if np.any(a_len > sys.maxsize / np.maximum(i, 1)):
205        raise OverflowError("Overflow encountered in string multiply")
206
207    buffersizes = a_len * i
208    out_dtype = f"{a.dtype.char}{buffersizes.max()}"
209    out = np.empty_like(a, shape=buffersizes.shape, dtype=out_dtype)
210    return _multiply_ufunc(a, i, out=out)
211
212
213def _mod_dispatcher(a, values):
214    return (a, values)
215
216
217@set_module("numpy.strings")
218@array_function_dispatch(_mod_dispatcher)
219def mod(a, values):
220    """
221    Return (a % i), that is pre-Python 2.6 string formatting
222    (interpolation), element-wise for a pair of array_likes of str
223    or unicode.
224
225    Parameters
226    ----------
227    a : array_like, with `np.bytes_` or `np.str_` dtype
228
229    values : array_like of values
230       These values will be element-wise interpolated into the string.
231
232    Returns
233    -------
234    out : ndarray
235        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
236        depending on input types
237
238    Examples
239    --------
240    >>> import numpy as np
241    >>> a = np.array(["NumPy is a %s library"])
242    >>> np.strings.mod(a, values=["Python"])
243    array(['NumPy is a Python library'], dtype='<U25')
244
245    >>> a = np.array([b'%d bytes', b'%d bits'])
246    >>> values = np.array([8, 64])
247    >>> np.strings.mod(a, values)
248    array([b'8 bytes', b'64 bits'], dtype='|S7')
249
250    """
251    return _to_bytes_or_str_array(
252        _vec_string(a, np.object_, '__mod__', (values,)), a)
253
254
255@set_module("numpy.strings")
256def find(a, sub, start=0, end=None):
257    """
258    For each element, return the lowest index in the string where
259    substring ``sub`` is found, such that ``sub`` is contained in the
260    range [``start``, ``end``).
261
262    Parameters
263    ----------
264    a : array_like, with ``StringDType``, ``bytes_`` or ``str_`` dtype
265
266    sub : array_like, with `np.bytes_` or `np.str_` dtype
267        The substring to search for.
268
269    start, end : array_like, with any integer dtype
270        The range to look in, interpreted as in slice notation.
271
272    Returns
273    -------
274    y : ndarray
275        Output array of ints
276
277    See Also
278    --------
279    str.find
280
281    Examples
282    --------
283    >>> import numpy as np
284    >>> a = np.array(["NumPy is a Python library"])
285    >>> np.strings.find(a, "Python")
286    array([11])
287
288    """
289    end = end if end is not None else MAX
290    return _find_ufunc(a, sub, start, end)
291
292
293@set_module("numpy.strings")
294def rfind(a, sub, start=0, end=None):
295    """
296    For each element, return the highest index in the string where
297    substring ``sub`` is found, such that ``sub`` is contained in the
298    range [``start``, ``end``).
299
300    Parameters
301    ----------
302    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
303
304    sub : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
305        The substring to search for.
306
307    start, end : array_like, with any integer dtype
308        The range to look in, interpreted as in slice notation.
309
310    Returns
311    -------
312    y : ndarray
313        Output array of ints
314
315    See Also
316    --------
317    str.rfind
318
319    Examples
320    --------
321    >>> import numpy as np
322    >>> a = np.array(["Computer Science"])
323    >>> np.strings.rfind(a, "Science", start=0, end=None)
324    array([9])
325    >>> np.strings.rfind(a, "Science", start=0, end=8)
326    array([-1])
327    >>> b = np.array(["Computer Science", "Science"])
328    >>> np.strings.rfind(b, "Science", start=0, end=None)
329    array([9, 0])
330
331    """
332    end = end if end is not None else MAX
333    return _rfind_ufunc(a, sub, start, end)
334
335
336@set_module("numpy.strings")
337def index(a, sub, start=0, end=None):
338    """
339    Like `find`, but raises :exc:`ValueError` when the substring is not found.
340
341    Parameters
342    ----------
343    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
344
345    sub : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
346
347    start, end : array_like, with any integer dtype, optional
348
349    Returns
350    -------
351    out : ndarray
352        Output array of ints.
353
354    See Also
355    --------
356    find, str.index
357
358    Examples
359    --------
360    >>> import numpy as np
361    >>> a = np.array(["Computer Science"])
362    >>> np.strings.index(a, "Science", start=0, end=None)
363    array([9])
364
365    """
366    end = end if end is not None else MAX
367    return _index_ufunc(a, sub, start, end)
368
369
370@set_module("numpy.strings")
371def rindex(a, sub, start=0, end=None):
372    """
373    Like `rfind`, but raises :exc:`ValueError` when the substring `sub` is
374    not found.
375
376    Parameters
377    ----------
378    a : array-like, with `np.bytes_` or `np.str_` dtype
379
380    sub : array-like, with `np.bytes_` or `np.str_` dtype
381
382    start, end : array-like, with any integer dtype, optional
383
384    Returns
385    -------
386    out : ndarray
387        Output array of ints.
388
389    See Also
390    --------
391    rfind, str.rindex
392
393    Examples
394    --------
395    >>> a = np.array(["Computer Science"])
396    >>> np.strings.rindex(a, "Science", start=0, end=None)
397    array([9])
398
399    """
400    end = end if end is not None else MAX
401    return _rindex_ufunc(a, sub, start, end)
402
403
404@set_module("numpy.strings")
405def count(a, sub, start=0, end=None):
406    """
407    Returns an array with the number of non-overlapping occurrences of
408    substring ``sub`` in the range [``start``, ``end``).
409
410    Parameters
411    ----------
412    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
413
414    sub : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
415       The substring to search for.
416
417    start, end : array_like, with any integer dtype
418        The range to look in, interpreted as in slice notation.
419
420    Returns
421    -------
422    y : ndarray
423        Output array of ints
424
425    See Also
426    --------
427    str.count
428
429    Examples
430    --------
431    >>> import numpy as np
432    >>> c = np.array(['aAaAaA', '  aA  ', 'abBABba'])
433    >>> c
434    array(['aAaAaA', '  aA  ', 'abBABba'], dtype='<U7')
435    >>> np.strings.count(c, 'A')
436    array([3, 1, 1])
437    >>> np.strings.count(c, 'aA')
438    array([3, 1, 0])
439    >>> np.strings.count(c, 'A', start=1, end=4)
440    array([2, 1, 1])
441    >>> np.strings.count(c, 'A', start=1, end=3)
442    array([1, 0, 0])
443
444    """
445    end = end if end is not None else MAX
446    return _count_ufunc(a, sub, start, end)
447
448
449@set_module("numpy.strings")
450def startswith(a, prefix, start=0, end=None):
451    """
452    Returns a boolean array which is `True` where the string element
453    in ``a`` starts with ``prefix``, otherwise `False`.
454
455    Parameters
456    ----------
457    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
458
459    prefix : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
460
461    start, end : array_like, with any integer dtype
462        With ``start``, test beginning at that position. With ``end``,
463        stop comparing at that position.
464
465    Returns
466    -------
467    out : ndarray
468        Output array of bools
469
470    See Also
471    --------
472    str.startswith
473
474    Examples
475    --------
476    >>> import numpy as np
477    >>> s = np.array(['foo', 'bar'])
478    >>> s
479    array(['foo', 'bar'], dtype='<U3')
480    >>> np.strings.startswith(s, 'fo')
481    array([True,  False])
482    >>> np.strings.startswith(s, 'o', start=1, end=2)
483    array([True,  False])
484
485    """
486    end = end if end is not None else MAX
487    return _startswith_ufunc(a, prefix, start, end)
488
489
490@set_module("numpy.strings")
491def endswith(a, suffix, start=0, end=None):
492    """
493    Returns a boolean array which is `True` where the string element
494    in ``a`` ends with ``suffix``, otherwise `False`.
495
496    Parameters
497    ----------
498    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
499
500    suffix : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
501
502    start, end : array_like, with any integer dtype
503        With ``start``, test beginning at that position. With ``end``,
504        stop comparing at that position.
505
506    Returns
507    -------
508    out : ndarray
509        Output array of bools
510
511    See Also
512    --------
513    str.endswith
514
515    Examples
516    --------
517    >>> import numpy as np
518    >>> s = np.array(['foo', 'bar'])
519    >>> s
520    array(['foo', 'bar'], dtype='<U3')
521    >>> np.strings.endswith(s, 'ar')
522    array([False,  True])
523    >>> np.strings.endswith(s, 'a', start=1, end=2)
524    array([False,  True])
525
526    """
527    end = end if end is not None else MAX
528    return _endswith_ufunc(a, suffix, start, end)
529
530
531def _code_dispatcher(a, encoding=None, errors=None):
532    return (a,)
533
534
535@set_module("numpy.strings")
536@array_function_dispatch(_code_dispatcher)
537def decode(a, encoding=None, errors=None):
538    r"""
539    Calls :meth:`bytes.decode` element-wise.
540
541    The set of available codecs comes from the Python standard library,
542    and may be extended at runtime.  For more information, see the
543    :mod:`codecs` module.
544
545    Parameters
546    ----------
547    a : array_like, with ``bytes_`` dtype
548
549    encoding : str, optional
550       The name of an encoding
551
552    errors : str, optional
553       Specifies how to handle encoding errors
554
555    Returns
556    -------
557    out : ndarray
558
559    See Also
560    --------
561    :py:meth:`bytes.decode`
562
563    Notes
564    -----
565    The type of the result will depend on the encoding specified.
566
567    Examples
568    --------
569    >>> import numpy as np
570    >>> c = np.array([b'\x81\xc1\x81\xc1\x81\xc1', b'@@\x81\xc1@@',
571    ...               b'\x81\x82\xc2\xc1\xc2\x82\x81'])
572    >>> c
573    array([b'\x81\xc1\x81\xc1\x81\xc1', b'@@\x81\xc1@@',
574           b'\x81\x82\xc2\xc1\xc2\x82\x81'], dtype='|S7')
575    >>> np.strings.decode(c, encoding='cp037')
576    array(['aAaAaA', '  aA  ', 'abBABba'], dtype='<U7')
577
578    """
579    return _to_bytes_or_str_array(
580        _vec_string(a, np.object_, 'decode', _clean_args(encoding, errors)),
581        np.str_(''))
582
583
584@set_module("numpy.strings")
585@array_function_dispatch(_code_dispatcher)
586def encode(a, encoding=None, errors=None):
587    """
588    Calls :meth:`str.encode` element-wise.
589
590    The set of available codecs comes from the Python standard library,
591    and may be extended at runtime. For more information, see the
592    :mod:`codecs` module.
593
594    Parameters
595    ----------
596    a : array_like, with ``StringDType`` or ``str_`` dtype
597
598    encoding : str, optional
599       The name of an encoding
600
601    errors : str, optional
602       Specifies how to handle encoding errors
603
604    Returns
605    -------
606    out : ndarray
607
608    See Also
609    --------
610    str.encode
611
612    Notes
613    -----
614    The type of the result will depend on the encoding specified.
615
616    Examples
617    --------
618    >>> import numpy as np
619    >>> a = np.array(['aAaAaA', '  aA  ', 'abBABba'])
620    >>> np.strings.encode(a, encoding='cp037')
621    array([b'\x81\xc1\x81\xc1\x81\xc1', b'@@\x81\xc1@@',
622       b'\x81\x82\xc2\xc1\xc2\x82\x81'], dtype='|S7')
623
624    """
625    return _to_bytes_or_str_array(
626        _vec_string(a, np.object_, 'encode', _clean_args(encoding, errors)),
627        np.bytes_(b''))
628
629
630def _expandtabs_dispatcher(a, tabsize=None):
631    return (a,)
632
633
634@set_module("numpy.strings")
635@array_function_dispatch(_expandtabs_dispatcher)
636def expandtabs(a, tabsize=8):
637    """
638    Return a copy of each string element where all tab characters are
639    replaced by one or more spaces.
640
641    Calls :meth:`str.expandtabs` element-wise.
642
643    Return a copy of each string element where all tab characters are
644    replaced by one or more spaces, depending on the current column
645    and the given `tabsize`. The column number is reset to zero after
646    each newline occurring in the string. This doesn't understand other
647    non-printing characters or escape sequences.
648
649    Parameters
650    ----------
651    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
652        Input array
653    tabsize : int, optional
654        Replace tabs with `tabsize` number of spaces.  If not given defaults
655        to 8 spaces.
656
657    Returns
658    -------
659    out : ndarray
660        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
661        depending on input type
662
663    See Also
664    --------
665    str.expandtabs
666
667    Examples
668    --------
669    >>> import numpy as np
670    >>> a = np.array(['\t\tHello\tworld'])
671    >>> np.strings.expandtabs(a, tabsize=4)  # doctest: +SKIP
672    array(['        Hello   world'], dtype='<U21')  # doctest: +SKIP
673
674    """
675    a = np.asanyarray(a)
676    tabsize = np.asanyarray(tabsize)
677
678    if a.dtype.char == "T":
679        return _expandtabs(a, tabsize)
680
681    buffersizes = _expandtabs_length(a, tabsize)
682    out_dtype = f"{a.dtype.char}{buffersizes.max()}"
683    out = np.empty_like(a, shape=buffersizes.shape, dtype=out_dtype)
684    return _expandtabs(a, tabsize, out=out)
685
686
687def _just_dispatcher(a, width, fillchar=None):
688    return (a,)
689
690
691@set_module("numpy.strings")
692@array_function_dispatch(_just_dispatcher)
693def center(a, width, fillchar=' '):
694    """
695    Return a copy of `a` with its elements centered in a string of
696    length `width`.
697
698    Parameters
699    ----------
700    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
701
702    width : array_like, with any integer dtype
703        The length of the resulting strings, unless ``width < str_len(a)``.
704    fillchar : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
705        Optional padding character to use (default is space).
706
707    Returns
708    -------
709    out : ndarray
710        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
711        depending on input types
712
713    See Also
714    --------
715    str.center
716
717    Notes
718    -----
719    While it is possible for ``a`` and ``fillchar`` to have different dtypes,
720    passing a non-ASCII character in ``fillchar`` when ``a`` is of dtype "S"
721    is not allowed, and a ``ValueError`` is raised.
722
723    Examples
724    --------
725    >>> import numpy as np
726    >>> c = np.array(['a1b2','1b2a','b2a1','2a1b']); c
727    array(['a1b2', '1b2a', 'b2a1', '2a1b'], dtype='<U4')
728    >>> np.strings.center(c, width=9)
729    array(['   a1b2  ', '   1b2a  ', '   b2a1  ', '   2a1b  '], dtype='<U9')
730    >>> np.strings.center(c, width=9, fillchar='*')
731    array(['***a1b2**', '***1b2a**', '***b2a1**', '***2a1b**'], dtype='<U9')
732    >>> np.strings.center(c, width=1)
733    array(['a1b2', '1b2a', 'b2a1', '2a1b'], dtype='<U4')
734
735    """
736    width = np.asanyarray(width)
737
738    if not np.issubdtype(width.dtype, np.integer):
739        raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
740
741    a = np.asanyarray(a)
742    fillchar = np.asanyarray(fillchar)
743
744    if np.any(str_len(fillchar) != 1):
745        raise TypeError(
746            "The fill character must be exactly one character long")
747
748    if np.result_type(a, fillchar).char == "T":
749        return _center(a, width, fillchar)
750
751    fillchar = fillchar.astype(a.dtype, copy=False)
752    width = np.maximum(str_len(a), width)
753    out_dtype = f"{a.dtype.char}{width.max()}"
754    shape = np.broadcast_shapes(a.shape, width.shape, fillchar.shape)
755    out = np.empty_like(a, shape=shape, dtype=out_dtype)
756
757    return _center(a, width, fillchar, out=out)
758
759
760@set_module("numpy.strings")
761@array_function_dispatch(_just_dispatcher)
762def ljust(a, width, fillchar=' '):
763    """
764    Return an array with the elements of `a` left-justified in a
765    string of length `width`.
766
767    Parameters
768    ----------
769    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
770
771    width : array_like, with any integer dtype
772        The length of the resulting strings, unless ``width < str_len(a)``.
773    fillchar : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
774        Optional character to use for padding (default is space).
775
776    Returns
777    -------
778    out : ndarray
779        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
780        depending on input types
781
782    See Also
783    --------
784    str.ljust
785
786    Notes
787    -----
788    While it is possible for ``a`` and ``fillchar`` to have different dtypes,
789    passing a non-ASCII character in ``fillchar`` when ``a`` is of dtype "S"
790    is not allowed, and a ``ValueError`` is raised.
791
792    Examples
793    --------
794    >>> import numpy as np
795    >>> c = np.array(['aAaAaA', '  aA  ', 'abBABba'])
796    >>> np.strings.ljust(c, width=3)
797    array(['aAaAaA', '  aA  ', 'abBABba'], dtype='<U7')
798    >>> np.strings.ljust(c, width=9)
799    array(['aAaAaA   ', '  aA     ', 'abBABba  '], dtype='<U9')
800
801    """
802    width = np.asanyarray(width)
803    if not np.issubdtype(width.dtype, np.integer):
804        raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
805
806    a = np.asanyarray(a)
807    fillchar = np.asanyarray(fillchar)
808
809    if np.any(str_len(fillchar) != 1):
810        raise TypeError(
811            "The fill character must be exactly one character long")
812
813    if np.result_type(a, fillchar).char == "T":
814        return _ljust(a, width, fillchar)
815
816    fillchar = fillchar.astype(a.dtype, copy=False)
817    width = np.maximum(str_len(a), width)
818    shape = np.broadcast_shapes(a.shape, width.shape, fillchar.shape)
819    out_dtype = f"{a.dtype.char}{width.max()}"
820    out = np.empty_like(a, shape=shape, dtype=out_dtype)
821
822    return _ljust(a, width, fillchar, out=out)
823
824
825@set_module("numpy.strings")
826@array_function_dispatch(_just_dispatcher)
827def rjust(a, width, fillchar=' '):
828    """
829    Return an array with the elements of `a` right-justified in a
830    string of length `width`.
831
832    Parameters
833    ----------
834    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
835
836    width : array_like, with any integer dtype
837        The length of the resulting strings, unless ``width < str_len(a)``.
838    fillchar : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
839        Optional padding character to use (default is space).
840
841    Returns
842    -------
843    out : ndarray
844        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
845        depending on input types
846
847    See Also
848    --------
849    str.rjust
850
851    Notes
852    -----
853    While it is possible for ``a`` and ``fillchar`` to have different dtypes,
854    passing a non-ASCII character in ``fillchar`` when ``a`` is of dtype "S"
855    is not allowed, and a ``ValueError`` is raised.
856
857    Examples
858    --------
859    >>> import numpy as np
860    >>> a = np.array(['aAaAaA', '  aA  ', 'abBABba'])
861    >>> np.strings.rjust(a, width=3)
862    array(['aAaAaA', '  aA  ', 'abBABba'], dtype='<U7')
863    >>> np.strings.rjust(a, width=9)
864    array(['   aAaAaA', '     aA  ', '  abBABba'], dtype='<U9')
865
866    """
867    width = np.asanyarray(width)
868    if not np.issubdtype(width.dtype, np.integer):
869        raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
870
871    a = np.asanyarray(a)
872    fillchar = np.asanyarray(fillchar)
873
874    if np.any(str_len(fillchar) != 1):
875        raise TypeError(
876            "The fill character must be exactly one character long")
877
878    if np.result_type(a, fillchar).char == "T":
879        return _rjust(a, width, fillchar)
880
881    fillchar = fillchar.astype(a.dtype, copy=False)
882    width = np.maximum(str_len(a), width)
883    shape = np.broadcast_shapes(a.shape, width.shape, fillchar.shape)
884    out_dtype = f"{a.dtype.char}{width.max()}"
885    out = np.empty_like(a, shape=shape, dtype=out_dtype)
886
887    return _rjust(a, width, fillchar, out=out)
888
889
890def _zfill_dispatcher(a, width):
891    return (a,)
892
893
894@set_module("numpy.strings")
895@array_function_dispatch(_zfill_dispatcher)
896def zfill(a, width):
897    """
898    Return the numeric string left-filled with zeros. A leading
899    sign prefix (``+``/``-``) is handled by inserting the padding
900    after the sign character rather than before.
901
902    Parameters
903    ----------
904    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
905
906    width : array_like, with any integer dtype
907        Width of string to left-fill elements in `a`.
908
909    Returns
910    -------
911    out : ndarray
912        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
913        depending on input type
914
915    See Also
916    --------
917    str.zfill
918
919    Examples
920    --------
921    >>> import numpy as np
922    >>> np.strings.zfill(['1', '-1', '+1'], 3)
923    array(['001', '-01', '+01'], dtype='<U3')
924
925    """
926    width = np.asanyarray(width)
927    if not np.issubdtype(width.dtype, np.integer):
928        raise TypeError(f"unsupported type {width.dtype} for operand 'width'")
929
930    a = np.asanyarray(a)
931
932    if a.dtype.char == "T":
933        return _zfill(a, width)
934
935    width = np.maximum(str_len(a), width)
936    shape = np.broadcast_shapes(a.shape, width.shape)
937    out_dtype = f"{a.dtype.char}{width.max()}"
938    out = np.empty_like(a, shape=shape, dtype=out_dtype)
939    return _zfill(a, width, out=out)
940
941
942@set_module("numpy.strings")
943def lstrip(a, chars=None):
944    """
945    For each element in `a`, return a copy with the leading characters
946    removed.
947
948    Parameters
949    ----------
950    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
951    chars : scalar with the same dtype as ``a``, optional
952       The ``chars`` argument is a string specifying the set of
953       characters to be removed. If ``None``, the ``chars``
954       argument defaults to removing whitespace. The ``chars`` argument
955       is not a prefix or suffix; rather, all combinations of its
956       values are stripped.
957
958    Returns
959    -------
960    out : ndarray
961        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
962        depending on input types
963
964    See Also
965    --------
966    str.lstrip
967
968    Examples
969    --------
970    >>> import numpy as np
971    >>> c = np.array(['aAaAaA', '  aA  ', 'abBABba'])
972    >>> c
973    array(['aAaAaA', '  aA  ', 'abBABba'], dtype='<U7')
974    # The 'a' variable is unstripped from c[1] because of leading whitespace.
975    >>> np.strings.lstrip(c, 'a')
976    array(['AaAaA', '  aA  ', 'bBABba'], dtype='<U7')
977    >>> np.strings.lstrip(c, 'A') # leaves c unchanged
978    array(['aAaAaA', '  aA  ', 'abBABba'], dtype='<U7')
979    >>> (np.strings.lstrip(c, ' ') == np.strings.lstrip(c, '')).all()
980    np.False_
981    >>> (np.strings.lstrip(c, ' ') == np.strings.lstrip(c)).all()
982    np.True_
983
984    """
985    if chars is None:
986        return _lstrip_whitespace(a)
987    return _lstrip_chars(a, chars)
988
989
990@set_module("numpy.strings")
991def rstrip(a, chars=None):
992    """
993    For each element in `a`, return a copy with the trailing characters
994    removed.
995
996    Parameters
997    ----------
998    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
999    chars : scalar with the same dtype as ``a``, optional
1000       The ``chars`` argument is a string specifying the set of
1001       characters to be removed. If ``None``, the ``chars``
1002       argument defaults to removing whitespace. The ``chars`` argument
1003       is not a prefix or suffix; rather, all combinations of its
1004       values are stripped.
1005
1006    Returns
1007    -------
1008    out : ndarray
1009        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1010        depending on input types
1011
1012    See Also
1013    --------
1014    str.rstrip
1015
1016    Examples
1017    --------
1018    >>> import numpy as np
1019    >>> c = np.array(['aAaAaA', 'abBABba'])
1020    >>> c
1021    array(['aAaAaA', 'abBABba'], dtype='<U7')
1022    >>> np.strings.rstrip(c, 'a')
1023    array(['aAaAaA', 'abBABb'], dtype='<U7')
1024    >>> np.strings.rstrip(c, 'A')
1025    array(['aAaAa', 'abBABba'], dtype='<U7')
1026
1027    """
1028    if chars is None:
1029        return _rstrip_whitespace(a)
1030    return _rstrip_chars(a, chars)
1031
1032
1033@set_module("numpy.strings")
1034def strip(a, chars=None):
1035    """
1036    For each element in `a`, return a copy with the leading and
1037    trailing characters removed.
1038
1039    Parameters
1040    ----------
1041    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1042    chars : scalar with the same dtype as ``a``, optional
1043       The ``chars`` argument is a string specifying the set of
1044       characters to be removed. If ``None``, the ``chars``
1045       argument defaults to removing whitespace. The ``chars`` argument
1046       is not a prefix or suffix; rather, all combinations of its
1047       values are stripped.
1048
1049    Returns
1050    -------
1051    out : ndarray
1052        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1053        depending on input types
1054
1055    See Also
1056    --------
1057    str.strip
1058
1059    Examples
1060    --------
1061    >>> import numpy as np
1062    >>> c = np.array(['aAaAaA', '  aA  ', 'abBABba'])
1063    >>> c
1064    array(['aAaAaA', '  aA  ', 'abBABba'], dtype='<U7')
1065    >>> np.strings.strip(c)
1066    array(['aAaAaA', 'aA', 'abBABba'], dtype='<U7')
1067    # 'a' unstripped from c[1] because of leading whitespace.
1068    >>> np.strings.strip(c, 'a')
1069    array(['AaAaA', '  aA  ', 'bBABb'], dtype='<U7')
1070    # 'A' unstripped from c[1] because of trailing whitespace.
1071    >>> np.strings.strip(c, 'A')
1072    array(['aAaAa', '  aA  ', 'abBABba'], dtype='<U7')
1073
1074    """
1075    if chars is None:
1076        return _strip_whitespace(a)
1077    return _strip_chars(a, chars)
1078
1079
1080def _unary_op_dispatcher(a):
1081    return (a,)
1082
1083
1084@set_module("numpy.strings")
1085@array_function_dispatch(_unary_op_dispatcher)
1086def upper(a):
1087    """
1088    Return an array with the elements converted to uppercase.
1089
1090    Calls :meth:`str.upper` element-wise.
1091
1092    For 8-bit strings, this method is locale-dependent.
1093
1094    Parameters
1095    ----------
1096    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1097        Input array.
1098
1099    Returns
1100    -------
1101    out : ndarray
1102        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1103        depending on input types
1104
1105    See Also
1106    --------
1107    str.upper
1108
1109    Examples
1110    --------
1111    >>> import numpy as np
1112    >>> c = np.array(['a1b c', '1bca', 'bca1']); c
1113    array(['a1b c', '1bca', 'bca1'], dtype='<U5')
1114    >>> np.strings.upper(c)
1115    array(['A1B C', '1BCA', 'BCA1'], dtype='<U5')
1116
1117    """
1118    a_arr = np.asarray(a)
1119    return _vec_string(a_arr, a_arr.dtype, 'upper')
1120
1121
1122@set_module("numpy.strings")
1123@array_function_dispatch(_unary_op_dispatcher)
1124def lower(a):
1125    """
1126    Return an array with the elements converted to lowercase.
1127
1128    Call :meth:`str.lower` element-wise.
1129
1130    For 8-bit strings, this method is locale-dependent.
1131
1132    Parameters
1133    ----------
1134    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1135        Input array.
1136
1137    Returns
1138    -------
1139    out : ndarray
1140        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1141        depending on input types
1142
1143    See Also
1144    --------
1145    str.lower
1146
1147    Examples
1148    --------
1149    >>> import numpy as np
1150    >>> c = np.array(['A1B C', '1BCA', 'BCA1']); c
1151    array(['A1B C', '1BCA', 'BCA1'], dtype='<U5')
1152    >>> np.strings.lower(c)
1153    array(['a1b c', '1bca', 'bca1'], dtype='<U5')
1154
1155    """
1156    a_arr = np.asarray(a)
1157    return _vec_string(a_arr, a_arr.dtype, 'lower')
1158
1159
1160@set_module("numpy.strings")
1161@array_function_dispatch(_unary_op_dispatcher)
1162def swapcase(a):
1163    """
1164    Return element-wise a copy of the string with
1165    uppercase characters converted to lowercase and vice versa.
1166
1167    Calls :meth:`str.swapcase` element-wise.
1168
1169    For 8-bit strings, this method is locale-dependent.
1170
1171    Parameters
1172    ----------
1173    a : array-like, with ``StringDType``, ``bytes_``, or ``str_`` dtype
1174        Input array.
1175
1176    Returns
1177    -------
1178    out : ndarray
1179        Output array of ``StringDType``, ``bytes_`` or ``str_`` dtype,
1180        depending on input types
1181
1182    See Also
1183    --------
1184    str.swapcase
1185
1186    Examples
1187    --------
1188    >>> import numpy as np
1189    >>> c=np.array(['a1B c','1b Ca','b Ca1','cA1b'],'S5'); c
1190    array(['a1B c', '1b Ca', 'b Ca1', 'cA1b'],
1191        dtype='|S5')
1192    >>> np.strings.swapcase(c)
1193    array(['A1b C', '1B cA', 'B cA1', 'Ca1B'],
1194        dtype='|S5')
1195
1196    """
1197    a_arr = np.asarray(a)
1198    return _vec_string(a_arr, a_arr.dtype, 'swapcase')
1199
1200

Showing the first 1,200 of 1814 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai