codekingpro/portable-devtools
115k
1// © 2016 and later: Unicode, Inc. and others.
2// License & terms of use: http://www.unicode.org/copyright.html
3/*
4**********************************************************************
5* Copyright (C) 1998-2016, International Business Machines
6* Corporation and others. All Rights Reserved.
7**********************************************************************
8*
9* File unistr.h
10*
11* Modification History:
12*
13* Date Name Description
14* 09/25/98 stephen Creation.
15* 11/11/98 stephen Changed per 11/9 code review.
16* 04/20/99 stephen Overhauled per 4/16 code review.
17* 11/18/99 aliu Made to inherit from Replaceable. Added method
18* handleReplaceBetween(); other methods unchanged.
19* 06/25/01 grhoten Remove dependency on iostream.
20******************************************************************************
21*/
22
23#ifndef UNISTR_H
24#define UNISTR_H
25
26/**
27 * \file
28 * \brief C++ API: Unicode String
29 */
30
31#include "unicode/utypes.h"
32
33#if U_SHOW_CPLUSPLUS_API
34
35#include <cstddef>
36#include "unicode/char16ptr.h"
37#include "unicode/rep.h"
38#include "unicode/std_string.h"
39#include "unicode/stringpiece.h"
40#include "unicode/bytestream.h"
41
42struct UConverter; // unicode/ucnv.h
43
44#ifndef USTRING_H
45/**
46 * \ingroup ustring_ustrlen
47 */
48U_STABLE int32_t U_EXPORT2
49u_strlen(const UChar *s);
50#endif
51
52U_NAMESPACE_BEGIN
53
54#if !UCONFIG_NO_BREAK_ITERATION
55class BreakIterator; // unicode/brkiter.h
56#endif
57class Edits;
58
59U_NAMESPACE_END
60
61// Not #ifndef U_HIDE_INTERNAL_API because UnicodeString needs the UStringCaseMapper.
62/**
63 * Internal string case mapping function type.
64 * All error checking must be done.
65 * src and dest must not overlap.
66 * @internal
67 */
68typedef int32_t U_CALLCONV
69UStringCaseMapper(int32_t caseLocale, uint32_t options,
70#if !UCONFIG_NO_BREAK_ITERATION
71 icu::BreakIterator *iter,
72#endif
73 char16_t *dest, int32_t destCapacity,
74 const char16_t *src, int32_t srcLength,
75 icu::Edits *edits,
76 UErrorCode &errorCode);
77
78U_NAMESPACE_BEGIN
79
80class Locale; // unicode/locid.h
81class StringCharacterIterator;
82class UnicodeStringAppendable; // unicode/appendable.h
83
84/* The <iostream> include has been moved to unicode/ustream.h */
85
86/**
87 * Constant to be used in the UnicodeString(char *, int32_t, EInvariant) constructor
88 * which constructs a Unicode string from an invariant-character char * string.
89 * About invariant characters see utypes.h.
90 * This constructor has no runtime dependency on conversion code and is
91 * therefore recommended over ones taking a charset name string
92 * (where the empty string "" indicates invariant-character conversion).
93 *
94 * @stable ICU 3.2
95 */
96#define US_INV icu::UnicodeString::kInvariant
97
98/**
99 * Unicode String literals in C++.
100 *
101 * Note: these macros are not recommended for new code.
102 * Prior to the availability of C++11 and u"unicode string literals",
103 * these macros were provided for portability and efficiency when
104 * initializing UnicodeStrings from literals.
105 *
106 * They work only for strings that contain "invariant characters", i.e.,
107 * only latin letters, digits, and some punctuation.
108 * See utypes.h for details.
109 *
110 * The string parameter must be a C string literal.
111 * The length of the string, not including the terminating
112 * `NUL`, must be specified as a constant.
113 * @stable ICU 2.0
114 */
115#if !U_CHAR16_IS_TYPEDEF
116# define UNICODE_STRING(cs, _length) icu::UnicodeString(TRUE, u ## cs, _length)
117#else
118# define UNICODE_STRING(cs, _length) icu::UnicodeString(TRUE, (const char16_t*)u ## cs, _length)
119#endif
120
121/**
122 * Unicode String literals in C++.
123 * Dependent on the platform properties, different UnicodeString
124 * constructors should be used to create a UnicodeString object from
125 * a string literal.
126 * The macros are defined for improved performance.
127 * They work only for strings that contain "invariant characters", i.e.,
128 * only latin letters, digits, and some punctuation.
129 * See utypes.h for details.
130 *
131 * The string parameter must be a C string literal.
132 * @stable ICU 2.0
133 */
134#define UNICODE_STRING_SIMPLE(cs) UNICODE_STRING(cs, -1)
135
136/**
137 * \def UNISTR_FROM_CHAR_EXPLICIT
138 * This can be defined to be empty or "explicit".
139 * If explicit, then the UnicodeString(char16_t) and UnicodeString(UChar32)
140 * constructors are marked as explicit, preventing their inadvertent use.
141 * @stable ICU 49
142 */
143#ifndef UNISTR_FROM_CHAR_EXPLICIT
144# if defined(U_COMBINED_IMPLEMENTATION) || defined(U_COMMON_IMPLEMENTATION) || defined(U_I18N_IMPLEMENTATION) || defined(U_IO_IMPLEMENTATION)
145 // Auto-"explicit" in ICU library code.
146# define UNISTR_FROM_CHAR_EXPLICIT explicit
147# else
148 // Empty by default for source code compatibility.
149# define UNISTR_FROM_CHAR_EXPLICIT
150# endif
151#endif
152
153/**
154 * \def UNISTR_FROM_STRING_EXPLICIT
155 * This can be defined to be empty or "explicit".
156 * If explicit, then the UnicodeString(const char *) and UnicodeString(const char16_t *)
157 * constructors are marked as explicit, preventing their inadvertent use.
158 *
159 * In particular, this helps prevent accidentally depending on ICU conversion code
160 * by passing a string literal into an API with a const UnicodeString & parameter.
161 * @stable ICU 49
162 */
163#ifndef UNISTR_FROM_STRING_EXPLICIT
164# if defined(U_COMBINED_IMPLEMENTATION) || defined(U_COMMON_IMPLEMENTATION) || defined(U_I18N_IMPLEMENTATION) || defined(U_IO_IMPLEMENTATION)
165 // Auto-"explicit" in ICU library code.
166# define UNISTR_FROM_STRING_EXPLICIT explicit
167# else
168 // Empty by default for source code compatibility.
169# define UNISTR_FROM_STRING_EXPLICIT
170# endif
171#endif
172
173/**
174 * \def UNISTR_OBJECT_SIZE
175 * Desired sizeof(UnicodeString) in bytes.
176 * It should be a multiple of sizeof(pointer) to avoid unusable space for padding.
177 * The object size may want to be a multiple of 16 bytes,
178 * which is a common granularity for heap allocation.
179 *
180 * Any space inside the object beyond sizeof(vtable pointer) + 2
181 * is available for storing short strings inside the object.
182 * The bigger the object, the longer a string that can be stored inside the object,
183 * without additional heap allocation.
184 *
185 * Depending on a platform's pointer size, pointer alignment requirements,
186 * and struct padding, the compiler will usually round up sizeof(UnicodeString)
187 * to 4 * sizeof(pointer) (or 3 * sizeof(pointer) for P128 data models),
188 * to hold the fields for heap-allocated strings.
189 * Such a minimum size also ensures that the object is easily large enough
190 * to hold at least 2 char16_ts, for one supplementary code point (U16_MAX_LENGTH).
191 *
192 * sizeof(UnicodeString) >= 48 should work for all known platforms.
193 *
194 * For example, on a 64-bit machine where sizeof(vtable pointer) is 8,
195 * sizeof(UnicodeString) = 64 would leave space for
196 * (64 - sizeof(vtable pointer) - 2) / U_SIZEOF_UCHAR = (64 - 8 - 2) / 2 = 27
197 * char16_ts stored inside the object.
198 *
199 * The minimum object size on a 64-bit machine would be
200 * 4 * sizeof(pointer) = 4 * 8 = 32 bytes,
201 * and the internal buffer would hold up to 11 char16_ts in that case.
202 *
203 * @see U16_MAX_LENGTH
204 * @stable ICU 56
205 */
206#ifndef UNISTR_OBJECT_SIZE
207# define UNISTR_OBJECT_SIZE 64
208#endif
209
210/**
211 * UnicodeString is a string class that stores Unicode characters directly and provides
212 * similar functionality as the Java String and StringBuffer/StringBuilder classes.
213 * It is a concrete implementation of the abstract class Replaceable (for transliteration).
214 *
215 * The UnicodeString equivalent of std::string’s clear() is remove().
216 *
217 * A UnicodeString may "alias" an external array of characters
218 * (that is, point to it, rather than own the array)
219 * whose lifetime must then at least match the lifetime of the aliasing object.
220 * This aliasing may be preserved when returning a UnicodeString by value,
221 * depending on the compiler and the function implementation,
222 * via Return Value Optimization (RVO) or the move assignment operator.
223 * (However, the copy assignment operator does not preserve aliasing.)
224 * For details see the description of storage models at the end of the class API docs
225 * and in the User Guide chapter linked from there.
226 *
227 * The UnicodeString class is not suitable for subclassing.
228 *
229 * For an overview of Unicode strings in C and C++ see the
230 * [User Guide Strings chapter](http://userguide.icu-project.org/strings#TOC-Strings-in-C-C-).
231 *
232 * In ICU, a Unicode string consists of 16-bit Unicode *code units*.
233 * A Unicode character may be stored with either one code unit
234 * (the most common case) or with a matched pair of special code units
235 * ("surrogates"). The data type for code units is char16_t.
236 * For single-character handling, a Unicode character code *point* is a value
237 * in the range 0..0x10ffff. ICU uses the UChar32 type for code points.
238 *
239 * Indexes and offsets into and lengths of strings always count code units, not code points.
240 * This is the same as with multi-byte char* strings in traditional string handling.
241 * Operations on partial strings typically do not test for code point boundaries.
242 * If necessary, the user needs to take care of such boundaries by testing for the code unit
243 * values or by using functions like
244 * UnicodeString::getChar32Start() and UnicodeString::getChar32Limit()
245 * (or, in C, the equivalent macros U16_SET_CP_START() and U16_SET_CP_LIMIT(), see utf.h).
246 *
247 * UnicodeString methods are more lenient with regard to input parameter values
248 * than other ICU APIs. In particular:
249 * - If indexes are out of bounds for a UnicodeString object
250 * (< 0 or > length()) then they are "pinned" to the nearest boundary.
251 * - If the buffer passed to an insert/append/replace operation is owned by the
252 * target object, e.g., calling str.append(str), an extra copy may take place
253 * to ensure safety.
254 * - If primitive string pointer values (e.g., const char16_t * or char *)
255 * for input strings are NULL, then those input string parameters are treated
256 * as if they pointed to an empty string.
257 * However, this is *not* the case for char * parameters for charset names
258 * or other IDs.
259 * - Most UnicodeString methods do not take a UErrorCode parameter because
260 * there are usually very few opportunities for failure other than a shortage
261 * of memory, error codes in low-level C++ string methods would be inconvenient,
262 * and the error code as the last parameter (ICU convention) would prevent
263 * the use of default parameter values.
264 * Instead, such methods set the UnicodeString into a "bogus" state
265 * (see isBogus()) if an error occurs.
266 *
267 * In string comparisons, two UnicodeString objects that are both "bogus"
268 * compare equal (to be transitive and prevent endless loops in sorting),
269 * and a "bogus" string compares less than any non-"bogus" one.
270 *
271 * Const UnicodeString methods are thread-safe. Multiple threads can use
272 * const methods on the same UnicodeString object simultaneously,
273 * but non-const methods must not be called concurrently (in multiple threads)
274 * with any other (const or non-const) methods.
275 *
276 * Similarly, const UnicodeString & parameters are thread-safe.
277 * One object may be passed in as such a parameter concurrently in multiple threads.
278 * This includes the const UnicodeString & parameters for
279 * copy construction, assignment, and cloning.
280 *
281 * UnicodeString uses several storage methods.
282 * String contents can be stored inside the UnicodeString object itself,
283 * in an allocated and shared buffer, or in an outside buffer that is "aliased".
284 * Most of this is done transparently, but careful aliasing in particular provides
285 * significant performance improvements.
286 * Also, the internal buffer is accessible via special functions.
287 * For details see the
288 * [User Guide Strings chapter](http://userguide.icu-project.org/strings#TOC-Maximizing-Performance-with-the-UnicodeString-Storage-Model).
289 *
290 * @see utf.h
291 * @see CharacterIterator
292 * @stable ICU 2.0
293 */
294class U_COMMON_API UnicodeString : public Replaceable
295{
296public:
297
298 /**
299 * Constant to be used in the UnicodeString(char *, int32_t, EInvariant) constructor
300 * which constructs a Unicode string from an invariant-character char * string.
301 * Use the macro US_INV instead of the full qualification for this value.
302 *
303 * @see US_INV
304 * @stable ICU 3.2
305 */
306 enum EInvariant {
307 /**
308 * @see EInvariant
309 * @stable ICU 3.2
310 */
311 kInvariant
312 };
313
314 //========================================
315 // Read-only operations
316 //========================================
317
318 /* Comparison - bitwise only - for international comparison use collation */
319
320 /**
321 * Equality operator. Performs only bitwise comparison.
322 * @param text The UnicodeString to compare to this one.
323 * @return TRUE if `text` contains the same characters as this one,
324 * FALSE otherwise.
325 * @stable ICU 2.0
326 */
327 inline UBool operator== (const UnicodeString& text) const;
328
329 /**
330 * Inequality operator. Performs only bitwise comparison.
331 * @param text The UnicodeString to compare to this one.
332 * @return FALSE if `text` contains the same characters as this one,
333 * TRUE otherwise.
334 * @stable ICU 2.0
335 */
336 inline UBool operator!= (const UnicodeString& text) const;
337
338 /**
339 * Greater than operator. Performs only bitwise comparison.
340 * @param text The UnicodeString to compare to this one.
341 * @return TRUE if the characters in this are bitwise
342 * greater than the characters in `text`, FALSE otherwise
343 * @stable ICU 2.0
344 */
345 inline UBool operator> (const UnicodeString& text) const;
346
347 /**
348 * Less than operator. Performs only bitwise comparison.
349 * @param text The UnicodeString to compare to this one.
350 * @return TRUE if the characters in this are bitwise
351 * less than the characters in `text`, FALSE otherwise
352 * @stable ICU 2.0
353 */
354 inline UBool operator< (const UnicodeString& text) const;
355
356 /**
357 * Greater than or equal operator. Performs only bitwise comparison.
358 * @param text The UnicodeString to compare to this one.
359 * @return TRUE if the characters in this are bitwise
360 * greater than or equal to the characters in `text`, FALSE otherwise
361 * @stable ICU 2.0
362 */
363 inline UBool operator>= (const UnicodeString& text) const;
364
365 /**
366 * Less than or equal operator. Performs only bitwise comparison.
367 * @param text The UnicodeString to compare to this one.
368 * @return TRUE if the characters in this are bitwise
369 * less than or equal to the characters in `text`, FALSE otherwise
370 * @stable ICU 2.0
371 */
372 inline UBool operator<= (const UnicodeString& text) const;
373
374 /**
375 * Compare the characters bitwise in this UnicodeString to
376 * the characters in `text`.
377 * @param text The UnicodeString to compare to this one.
378 * @return The result of bitwise character comparison: 0 if this
379 * contains the same characters as `text`, -1 if the characters in
380 * this are bitwise less than the characters in `text`, +1 if the
381 * characters in this are bitwise greater than the characters
382 * in `text`.
383 * @stable ICU 2.0
384 */
385 inline int8_t compare(const UnicodeString& text) const;
386
387 /**
388 * Compare the characters bitwise in the range
389 * [`start`, `start + length`) with the characters
390 * in the **entire string** `text`.
391 * (The parameters "start" and "length" are not applied to the other text "text".)
392 * @param start the offset at which the compare operation begins
393 * @param length the number of characters of text to compare.
394 * @param text the other text to be compared against this string.
395 * @return The result of bitwise character comparison: 0 if this
396 * contains the same characters as `text`, -1 if the characters in
397 * this are bitwise less than the characters in `text`, +1 if the
398 * characters in this are bitwise greater than the characters
399 * in `text`.
400 * @stable ICU 2.0
401 */
402 inline int8_t compare(int32_t start,
403 int32_t length,
404 const UnicodeString& text) const;
405
406 /**
407 * Compare the characters bitwise in the range
408 * [`start`, `start + length`) with the characters
409 * in `srcText` in the range
410 * [`srcStart`, `srcStart + srcLength`).
411 * @param start the offset at which the compare operation begins
412 * @param length the number of characters in this to compare.
413 * @param srcText the text to be compared
414 * @param srcStart the offset into `srcText` to start comparison
415 * @param srcLength the number of characters in `src` to compare
416 * @return The result of bitwise character comparison: 0 if this
417 * contains the same characters as `srcText`, -1 if the characters in
418 * this are bitwise less than the characters in `srcText`, +1 if the
419 * characters in this are bitwise greater than the characters
420 * in `srcText`.
421 * @stable ICU 2.0
422 */
423 inline int8_t compare(int32_t start,
424 int32_t length,
425 const UnicodeString& srcText,
426 int32_t srcStart,
427 int32_t srcLength) const;
428
429 /**
430 * Compare the characters bitwise in this UnicodeString with the first
431 * `srcLength` characters in `srcChars`.
432 * @param srcChars The characters to compare to this UnicodeString.
433 * @param srcLength the number of characters in `srcChars` to compare
434 * @return The result of bitwise character comparison: 0 if this
435 * contains the same characters as `srcChars`, -1 if the characters in
436 * this are bitwise less than the characters in `srcChars`, +1 if the
437 * characters in this are bitwise greater than the characters
438 * in `srcChars`.
439 * @stable ICU 2.0
440 */
441 inline int8_t compare(ConstChar16Ptr srcChars,
442 int32_t srcLength) const;
443
444 /**
445 * Compare the characters bitwise in the range
446 * [`start`, `start + length`) with the first
447 * `length` characters in `srcChars`
448 * @param start the offset at which the compare operation begins
449 * @param length the number of characters to compare.
450 * @param srcChars the characters to be compared
451 * @return The result of bitwise character comparison: 0 if this
452 * contains the same characters as `srcChars`, -1 if the characters in
453 * this are bitwise less than the characters in `srcChars`, +1 if the
454 * characters in this are bitwise greater than the characters
455 * in `srcChars`.
456 * @stable ICU 2.0
457 */
458 inline int8_t compare(int32_t start,
459 int32_t length,
460 const char16_t *srcChars) const;
461
462 /**
463 * Compare the characters bitwise in the range
464 * [`start`, `start + length`) with the characters
465 * in `srcChars` in the range
466 * [`srcStart`, `srcStart + srcLength`).
467 * @param start the offset at which the compare operation begins
468 * @param length the number of characters in this to compare
469 * @param srcChars the characters to be compared
470 * @param srcStart the offset into `srcChars` to start comparison
471 * @param srcLength the number of characters in `srcChars` to compare
472 * @return The result of bitwise character comparison: 0 if this
473 * contains the same characters as `srcChars`, -1 if the characters in
474 * this are bitwise less than the characters in `srcChars`, +1 if the
475 * characters in this are bitwise greater than the characters
476 * in `srcChars`.
477 * @stable ICU 2.0
478 */
479 inline int8_t compare(int32_t start,
480 int32_t length,
481 const char16_t *srcChars,
482 int32_t srcStart,
483 int32_t srcLength) const;
484
485 /**
486 * Compare the characters bitwise in the range
487 * [`start`, `limit`) with the characters
488 * in `srcText` in the range
489 * [`srcStart`, `srcLimit`).
490 * @param start the offset at which the compare operation begins
491 * @param limit the offset immediately following the compare operation
492 * @param srcText the text to be compared
493 * @param srcStart the offset into `srcText` to start comparison
494 * @param srcLimit the offset into `srcText` to limit comparison
495 * @return The result of bitwise character comparison: 0 if this
496 * contains the same characters as `srcText`, -1 if the characters in
497 * this are bitwise less than the characters in `srcText`, +1 if the
498 * characters in this are bitwise greater than the characters
499 * in `srcText`.
500 * @stable ICU 2.0
501 */
502 inline int8_t compareBetween(int32_t start,
503 int32_t limit,
504 const UnicodeString& srcText,
505 int32_t srcStart,
506 int32_t srcLimit) const;
507
508 /**
509 * Compare two Unicode strings in code point order.
510 * The result may be different from the results of compare(), operator<, etc.
511 * if supplementary characters are present:
512 *
513 * In UTF-16, supplementary characters (with code points U+10000 and above) are
514 * stored with pairs of surrogate code units. These have values from 0xd800 to 0xdfff,
515 * which means that they compare as less than some other BMP characters like U+feff.
516 * This function compares Unicode strings in code point order.
517 * If either of the UTF-16 strings is malformed (i.e., it contains unpaired surrogates), then the result is not defined.
518 *
519 * @param text Another string to compare this one to.
520 * @return a negative/zero/positive integer corresponding to whether
521 * this string is less than/equal to/greater than the second one
522 * in code point order
523 * @stable ICU 2.0
524 */
525 inline int8_t compareCodePointOrder(const UnicodeString& text) const;
526
527 /**
528 * Compare two Unicode strings in code point order.
529 * The result may be different from the results of compare(), operator<, etc.
530 * if supplementary characters are present:
531 *
532 * In UTF-16, supplementary characters (with code points U+10000 and above) are
533 * stored with pairs of surrogate code units. These have values from 0xd800 to 0xdfff,
534 * which means that they compare as less than some other BMP characters like U+feff.
535 * This function compares Unicode strings in code point order.
536 * If either of the UTF-16 strings is malformed (i.e., it contains unpaired surrogates), then the result is not defined.
537 *
538 * @param start The start offset in this string at which the compare operation begins.
539 * @param length The number of code units from this string to compare.
540 * @param srcText Another string to compare this one to.
541 * @return a negative/zero/positive integer corresponding to whether
542 * this string is less than/equal to/greater than the second one
543 * in code point order
544 * @stable ICU 2.0
545 */
546 inline int8_t compareCodePointOrder(int32_t start,
547 int32_t length,
548 const UnicodeString& srcText) const;
549
550 /**
551 * Compare two Unicode strings in code point order.
552 * The result may be different from the results of compare(), operator<, etc.
553 * if supplementary characters are present:
554 *
555 * In UTF-16, supplementary characters (with code points U+10000 and above) are
556 * stored with pairs of surrogate code units. These have values from 0xd800 to 0xdfff,
557 * which means that they compare as less than some other BMP characters like U+feff.
558 * This function compares Unicode strings in code point order.
559 * If either of the UTF-16 strings is malformed (i.e., it contains unpaired surrogates), then the result is not defined.
560 *
561 * @param start The start offset in this string at which the compare operation begins.
562 * @param length The number of code units from this string to compare.
563 * @param srcText Another string to compare this one to.
564 * @param srcStart The start offset in that string at which the compare operation begins.
565 * @param srcLength The number of code units from that string to compare.
566 * @return a negative/zero/positive integer corresponding to whether
567 * this string is less than/equal to/greater than the second one
568 * in code point order
569 * @stable ICU 2.0
570 */
571 inline int8_t compareCodePointOrder(int32_t start,
572 int32_t length,
573 const UnicodeString& srcText,
574 int32_t srcStart,
575 int32_t srcLength) const;
576
577 /**
578 * Compare two Unicode strings in code point order.
579 * The result may be different from the results of compare(), operator<, etc.
580 * if supplementary characters are present:
581 *
582 * In UTF-16, supplementary characters (with code points U+10000 and above) are
583 * stored with pairs of surrogate code units. These have values from 0xd800 to 0xdfff,
584 * which means that they compare as less than some other BMP characters like U+feff.
585 * This function compares Unicode strings in code point order.
586 * If either of the UTF-16 strings is malformed (i.e., it contains unpaired surrogates), then the result is not defined.
587 *
588 * @param srcChars A pointer to another string to compare this one to.
589 * @param srcLength The number of code units from that string to compare.
590 * @return a negative/zero/positive integer corresponding to whether
591 * this string is less than/equal to/greater than the second one
592 * in code point order
593 * @stable ICU 2.0
594 */
595 inline int8_t compareCodePointOrder(ConstChar16Ptr srcChars,
596 int32_t srcLength) const;
597
598 /**
599 * Compare two Unicode strings in code point order.
600 * The result may be different from the results of compare(), operator<, etc.
601 * if supplementary characters are present:
602 *
603 * In UTF-16, supplementary characters (with code points U+10000 and above) are
604 * stored with pairs of surrogate code units. These have values from 0xd800 to 0xdfff,
605 * which means that they compare as less than some other BMP characters like U+feff.
606 * This function compares Unicode strings in code point order.
607 * If either of the UTF-16 strings is malformed (i.e., it contains unpaired surrogates), then the result is not defined.
608 *
609 * @param start The start offset in this string at which the compare operation begins.
610 * @param length The number of code units from this string to compare.
611 * @param srcChars A pointer to another string to compare this one to.
612 * @return a negative/zero/positive integer corresponding to whether
613 * this string is less than/equal to/greater than the second one
614 * in code point order
615 * @stable ICU 2.0
616 */
617 inline int8_t compareCodePointOrder(int32_t start,
618 int32_t length,
619 const char16_t *srcChars) const;
620
621 /**
622 * Compare two Unicode strings in code point order.
623 * The result may be different from the results of compare(), operator<, etc.
624 * if supplementary characters are present:
625 *
626 * In UTF-16, supplementary characters (with code points U+10000 and above) are
627 * stored with pairs of surrogate code units. These have values from 0xd800 to 0xdfff,
628 * which means that they compare as less than some other BMP characters like U+feff.
629 * This function compares Unicode strings in code point order.
630 * If either of the UTF-16 strings is malformed (i.e., it contains unpaired surrogates), then the result is not defined.
631 *
632 * @param start The start offset in this string at which the compare operation begins.
633 * @param length The number of code units from this string to compare.
634 * @param srcChars A pointer to another string to compare this one to.
635 * @param srcStart The start offset in that string at which the compare operation begins.
636 * @param srcLength The number of code units from that string to compare.
637 * @return a negative/zero/positive integer corresponding to whether
638 * this string is less than/equal to/greater than the second one
639 * in code point order
640 * @stable ICU 2.0
641 */
642 inline int8_t compareCodePointOrder(int32_t start,
643 int32_t length,
644 const char16_t *srcChars,
645 int32_t srcStart,
646 int32_t srcLength) const;
647
648 /**
649 * Compare two Unicode strings in code point order.
650 * The result may be different from the results of compare(), operator<, etc.
651 * if supplementary characters are present:
652 *
653 * In UTF-16, supplementary characters (with code points U+10000 and above) are
654 * stored with pairs of surrogate code units. These have values from 0xd800 to 0xdfff,
655 * which means that they compare as less than some other BMP characters like U+feff.
656 * This function compares Unicode strings in code point order.
657 * If either of the UTF-16 strings is malformed (i.e., it contains unpaired surrogates), then the result is not defined.
658 *
659 * @param start The start offset in this string at which the compare operation begins.
660 * @param limit The offset after the last code unit from this string to compare.
661 * @param srcText Another string to compare this one to.
662 * @param srcStart The start offset in that string at which the compare operation begins.
663 * @param srcLimit The offset after the last code unit from that string to compare.
664 * @return a negative/zero/positive integer corresponding to whether
665 * this string is less than/equal to/greater than the second one
666 * in code point order
667 * @stable ICU 2.0
668 */
669 inline int8_t compareCodePointOrderBetween(int32_t start,
670 int32_t limit,
671 const UnicodeString& srcText,
672 int32_t srcStart,
673 int32_t srcLimit) const;
674
675 /**
676 * Compare two strings case-insensitively using full case folding.
677 * This is equivalent to this->foldCase(options).compare(text.foldCase(options)).
678 *
679 * @param text Another string to compare this one to.
680 * @param options A bit set of options:
681 * - U_FOLD_CASE_DEFAULT or 0 is used for default options:
682 * Comparison in code unit order with default case folding.
683 *
684 * - U_COMPARE_CODE_POINT_ORDER
685 * Set to choose code point order instead of code unit order
686 * (see u_strCompare for details).
687 *
688 * - U_FOLD_CASE_EXCLUDE_SPECIAL_I
689 *
690 * @return A negative, zero, or positive integer indicating the comparison result.
691 * @stable ICU 2.0
692 */
693 inline int8_t caseCompare(const UnicodeString& text, uint32_t options) const;
694
695 /**
696 * Compare two strings case-insensitively using full case folding.
697 * This is equivalent to this->foldCase(options).compare(srcText.foldCase(options)).
698 *
699 * @param start The start offset in this string at which the compare operation begins.
700 * @param length The number of code units from this string to compare.
701 * @param srcText Another string to compare this one to.
702 * @param options A bit set of options:
703 * - U_FOLD_CASE_DEFAULT or 0 is used for default options:
704 * Comparison in code unit order with default case folding.
705 *
706 * - U_COMPARE_CODE_POINT_ORDER
707 * Set to choose code point order instead of code unit order
708 * (see u_strCompare for details).
709 *
710 * - U_FOLD_CASE_EXCLUDE_SPECIAL_I
711 *
712 * @return A negative, zero, or positive integer indicating the comparison result.
713 * @stable ICU 2.0
714 */
715 inline int8_t caseCompare(int32_t start,
716 int32_t length,
717 const UnicodeString& srcText,
718 uint32_t options) const;
719
720 /**
721 * Compare two strings case-insensitively using full case folding.
722 * This is equivalent to this->foldCase(options).compare(srcText.foldCase(options)).
723 *
724 * @param start The start offset in this string at which the compare operation begins.
725 * @param length The number of code units from this string to compare.
726 * @param srcText Another string to compare this one to.
727 * @param srcStart The start offset in that string at which the compare operation begins.
728 * @param srcLength The number of code units from that string to compare.
729 * @param options A bit set of options:
730 * - U_FOLD_CASE_DEFAULT or 0 is used for default options:
731 * Comparison in code unit order with default case folding.
732 *
733 * - U_COMPARE_CODE_POINT_ORDER
734 * Set to choose code point order instead of code unit order
735 * (see u_strCompare for details).
736 *
737 * - U_FOLD_CASE_EXCLUDE_SPECIAL_I
738 *
739 * @return A negative, zero, or positive integer indicating the comparison result.
740 * @stable ICU 2.0
741 */
742 inline int8_t caseCompare(int32_t start,
743 int32_t length,
744 const UnicodeString& srcText,
745 int32_t srcStart,
746 int32_t srcLength,
747 uint32_t options) const;
748
749 /**
750 * Compare two strings case-insensitively using full case folding.
751 * This is equivalent to this->foldCase(options).compare(srcChars.foldCase(options)).
752 *
753 * @param srcChars A pointer to another string to compare this one to.
754 * @param srcLength The number of code units from that string to compare.
755 * @param options A bit set of options:
756 * - U_FOLD_CASE_DEFAULT or 0 is used for default options:
757 * Comparison in code unit order with default case folding.
758 *
759 * - U_COMPARE_CODE_POINT_ORDER
760 * Set to choose code point order instead of code unit order
761 * (see u_strCompare for details).
762 *
763 * - U_FOLD_CASE_EXCLUDE_SPECIAL_I
764 *
765 * @return A negative, zero, or positive integer indicating the comparison result.
766 * @stable ICU 2.0
767 */
768 inline int8_t caseCompare(ConstChar16Ptr srcChars,
769 int32_t srcLength,
770 uint32_t options) const;
771
772 /**
773 * Compare two strings case-insensitively using full case folding.
774 * This is equivalent to this->foldCase(options).compare(srcChars.foldCase(options)).
775 *
776 * @param start The start offset in this string at which the compare operation begins.
777 * @param length The number of code units from this string to compare.
778 * @param srcChars A pointer to another string to compare this one to.
779 * @param options A bit set of options:
780 * - U_FOLD_CASE_DEFAULT or 0 is used for default options:
781 * Comparison in code unit order with default case folding.
782 *
783 * - U_COMPARE_CODE_POINT_ORDER
784 * Set to choose code point order instead of code unit order
785 * (see u_strCompare for details).
786 *
787 * - U_FOLD_CASE_EXCLUDE_SPECIAL_I
788 *
789 * @return A negative, zero, or positive integer indicating the comparison result.
790 * @stable ICU 2.0
791 */
792 inline int8_t caseCompare(int32_t start,
793 int32_t length,
794 const char16_t *srcChars,
795 uint32_t options) const;
796
797 /**
798 * Compare two strings case-insensitively using full case folding.
799 * This is equivalent to this->foldCase(options).compare(srcChars.foldCase(options)).
800 *
801 * @param start The start offset in this string at which the compare operation begins.
802 * @param length The number of code units from this string to compare.
803 * @param srcChars A pointer to another string to compare this one to.
804 * @param srcStart The start offset in that string at which the compare operation begins.
805 * @param srcLength The number of code units from that string to compare.
806 * @param options A bit set of options:
807 * - U_FOLD_CASE_DEFAULT or 0 is used for default options:
808 * Comparison in code unit order with default case folding.
809 *
810 * - U_COMPARE_CODE_POINT_ORDER
811 * Set to choose code point order instead of code unit order
812 * (see u_strCompare for details).
813 *
814 * - U_FOLD_CASE_EXCLUDE_SPECIAL_I
815 *
816 * @return A negative, zero, or positive integer indicating the comparison result.
817 * @stable ICU 2.0
818 */
819 inline int8_t caseCompare(int32_t start,
820 int32_t length,
821 const char16_t *srcChars,
822 int32_t srcStart,
823 int32_t srcLength,
824 uint32_t options) const;
825
826 /**
827 * Compare two strings case-insensitively using full case folding.
828 * This is equivalent to this->foldCase(options).compareBetween(text.foldCase(options)).
829 *
830 * @param start The start offset in this string at which the compare operation begins.
831 * @param limit The offset after the last code unit from this string to compare.
832 * @param srcText Another string to compare this one to.
833 * @param srcStart The start offset in that string at which the compare operation begins.
834 * @param srcLimit The offset after the last code unit from that string to compare.
835 * @param options A bit set of options:
836 * - U_FOLD_CASE_DEFAULT or 0 is used for default options:
837 * Comparison in code unit order with default case folding.
838 *
839 * - U_COMPARE_CODE_POINT_ORDER
840 * Set to choose code point order instead of code unit order
841 * (see u_strCompare for details).
842 *
843 * - U_FOLD_CASE_EXCLUDE_SPECIAL_I
844 *
845 * @return A negative, zero, or positive integer indicating the comparison result.
846 * @stable ICU 2.0
847 */
848 inline int8_t caseCompareBetween(int32_t start,
849 int32_t limit,
850 const UnicodeString& srcText,
851 int32_t srcStart,
852 int32_t srcLimit,
853 uint32_t options) const;
854
855 /**
856 * Determine if this starts with the characters in `text`
857 * @param text The text to match.
858 * @return TRUE if this starts with the characters in `text`,
859 * FALSE otherwise
860 * @stable ICU 2.0
861 */
862 inline UBool startsWith(const UnicodeString& text) const;
863
864 /**
865 * Determine if this starts with the characters in `srcText`
866 * in the range [`srcStart`, `srcStart + srcLength`).
867 * @param srcText The text to match.
868 * @param srcStart the offset into `srcText` to start matching
869 * @param srcLength the number of characters in `srcText` to match
870 * @return TRUE if this starts with the characters in `text`,
871 * FALSE otherwise
872 * @stable ICU 2.0
873 */
874 inline UBool startsWith(const UnicodeString& srcText,
875 int32_t srcStart,
876 int32_t srcLength) const;
877
878 /**
879 * Determine if this starts with the characters in `srcChars`
880 * @param srcChars The characters to match.
881 * @param srcLength the number of characters in `srcChars`
882 * @return TRUE if this starts with the characters in `srcChars`,
883 * FALSE otherwise
884 * @stable ICU 2.0
885 */
886 inline UBool startsWith(ConstChar16Ptr srcChars,
887 int32_t srcLength) const;
888
889 /**
890 * Determine if this ends with the characters in `srcChars`
891 * in the range [`srcStart`, `srcStart + srcLength`).
892 * @param srcChars The characters to match.
893 * @param srcStart the offset into `srcText` to start matching
894 * @param srcLength the number of characters in `srcChars` to match
895 * @return TRUE if this ends with the characters in `srcChars`, FALSE otherwise
896 * @stable ICU 2.0
897 */
898 inline UBool startsWith(const char16_t *srcChars,
899 int32_t srcStart,
900 int32_t srcLength) const;
901
902 /**
903 * Determine if this ends with the characters in `text`
904 * @param text The text to match.
905 * @return TRUE if this ends with the characters in `text`,
906 * FALSE otherwise
907 * @stable ICU 2.0
908 */
909 inline UBool endsWith(const UnicodeString& text) const;
910
911 /**
912 * Determine if this ends with the characters in `srcText`
913 * in the range [`srcStart`, `srcStart + srcLength`).
914 * @param srcText The text to match.
915 * @param srcStart the offset into `srcText` to start matching
916 * @param srcLength the number of characters in `srcText` to match
917 * @return TRUE if this ends with the characters in `text`,
918 * FALSE otherwise
919 * @stable ICU 2.0
920 */
921 inline UBool endsWith(const UnicodeString& srcText,
922 int32_t srcStart,
923 int32_t srcLength) const;
924
925 /**
926 * Determine if this ends with the characters in `srcChars`
927 * @param srcChars The characters to match.
928 * @param srcLength the number of characters in `srcChars`
929 * @return TRUE if this ends with the characters in `srcChars`,
930 * FALSE otherwise
931 * @stable ICU 2.0
932 */
933 inline UBool endsWith(ConstChar16Ptr srcChars,
934 int32_t srcLength) const;
935
936 /**
937 * Determine if this ends with the characters in `srcChars`
938 * in the range [`srcStart`, `srcStart + srcLength`).
939 * @param srcChars The characters to match.
940 * @param srcStart the offset into `srcText` to start matching
941 * @param srcLength the number of characters in `srcChars` to match
942 * @return TRUE if this ends with the characters in `srcChars`,
943 * FALSE otherwise
944 * @stable ICU 2.0
945 */
946 inline UBool endsWith(const char16_t *srcChars,
947 int32_t srcStart,
948 int32_t srcLength) const;
949
950
951 /* Searching - bitwise only */
952
953 /**
954 * Locate in this the first occurrence of the characters in `text`,
955 * using bitwise comparison.
956 * @param text The text to search for.
957 * @return The offset into this of the start of `text`,
958 * or -1 if not found.
959 * @stable ICU 2.0
960 */
961 inline int32_t indexOf(const UnicodeString& text) const;
962
963 /**
964 * Locate in this the first occurrence of the characters in `text`
965 * starting at offset `start`, using bitwise comparison.
966 * @param text The text to search for.
967 * @param start The offset at which searching will start.
968 * @return The offset into this of the start of `text`,
969 * or -1 if not found.
970 * @stable ICU 2.0
971 */
972 inline int32_t indexOf(const UnicodeString& text,
973 int32_t start) const;
974
975 /**
976 * Locate in this the first occurrence in the range
977 * [`start`, `start + length`) of the characters
978 * in `text`, using bitwise comparison.
979 * @param text The text to search for.
980 * @param start The offset at which searching will start.
981 * @param length The number of characters to search
982 * @return The offset into this of the start of `text`,
983 * or -1 if not found.
984 * @stable ICU 2.0
985 */
986 inline int32_t indexOf(const UnicodeString& text,
987 int32_t start,
988 int32_t length) const;
989
990 /**
991 * Locate in this the first occurrence in the range
992 * [`start`, `start + length`) of the characters
993 * in `srcText` in the range
994 * [`srcStart`, `srcStart + srcLength`),
995 * using bitwise comparison.
996 * @param srcText The text to search for.
997 * @param srcStart the offset into `srcText` at which
998 * to start matching
999 * @param srcLength the number of characters in `srcText` to match
1000 * @param start the offset into this at which to start matching
1001 * @param length the number of characters in this to search
1002 * @return The offset into this of the start of `text`,
1003 * or -1 if not found.
1004 * @stable ICU 2.0
1005 */
1006 inline int32_t indexOf(const UnicodeString& srcText,
1007 int32_t srcStart,
1008 int32_t srcLength,
1009 int32_t start,
1010 int32_t length) const;
1011
1012 /**
1013 * Locate in this the first occurrence of the characters in
1014 * `srcChars`
1015 * starting at offset `start`, using bitwise comparison.
1016 * @param srcChars The text to search for.
1017 * @param srcLength the number of characters in `srcChars` to match
1018 * @param start the offset into this at which to start matching
1019 * @return The offset into this of the start of `text`,
1020 * or -1 if not found.
1021 * @stable ICU 2.0
1022 */
1023 inline int32_t indexOf(const char16_t *srcChars,
1024 int32_t srcLength,
1025 int32_t start) const;
1026
1027 /**
1028 * Locate in this the first occurrence in the range
1029 * [`start`, `start + length`) of the characters
1030 * in `srcChars`, using bitwise comparison.
1031 * @param srcChars The text to search for.
1032 * @param srcLength the number of characters in `srcChars`
1033 * @param start The offset at which searching will start.
1034 * @param length The number of characters to search
1035 * @return The offset into this of the start of `srcChars`,
1036 * or -1 if not found.
1037 * @stable ICU 2.0
1038 */
1039 inline int32_t indexOf(ConstChar16Ptr srcChars,
1040 int32_t srcLength,
1041 int32_t start,
1042 int32_t length) const;
1043
1044 /**
1045 * Locate in this the first occurrence in the range
1046 * [`start`, `start + length`) of the characters
1047 * in `srcChars` in the range
1048 * [`srcStart`, `srcStart + srcLength`),
1049 * using bitwise comparison.
1050 * @param srcChars The text to search for.
1051 * @param srcStart the offset into `srcChars` at which
1052 * to start matching
1053 * @param srcLength the number of characters in `srcChars` to match
1054 * @param start the offset into this at which to start matching
1055 * @param length the number of characters in this to search
1056 * @return The offset into this of the start of `text`,
1057 * or -1 if not found.
1058 * @stable ICU 2.0
1059 */
1060 int32_t indexOf(const char16_t *srcChars,
1061 int32_t srcStart,
1062 int32_t srcLength,
1063 int32_t start,
1064 int32_t length) const;
1065
1066 /**
1067 * Locate in this the first occurrence of the BMP code point `c`,
1068 * using bitwise comparison.
1069 * @param c The code unit to search for.
1070 * @return The offset into this of `c`, or -1 if not found.
1071 * @stable ICU 2.0
1072 */
1073 inline int32_t indexOf(char16_t c) const;
1074
1075 /**
1076 * Locate in this the first occurrence of the code point `c`,
1077 * using bitwise comparison.
1078 *
1079 * @param c The code point to search for.
1080 * @return The offset into this of `c`, or -1 if not found.
1081 * @stable ICU 2.0
1082 */
1083 inline int32_t indexOf(UChar32 c) const;
1084
1085 /**
1086 * Locate in this the first occurrence of the BMP code point `c`,
1087 * starting at offset `start`, using bitwise comparison.
1088 * @param c The code unit to search for.
1089 * @param start The offset at which searching will start.
1090 * @return The offset into this of `c`, or -1 if not found.
1091 * @stable ICU 2.0
1092 */
1093 inline int32_t indexOf(char16_t c,
1094 int32_t start) const;
1095
1096 /**
1097 * Locate in this the first occurrence of the code point `c`
1098 * starting at offset `start`, using bitwise comparison.
1099 *
1100 * @param c The code point to search for.
1101 * @param start The offset at which searching will start.
1102 * @return The offset into this of `c`, or -1 if not found.
1103 * @stable ICU 2.0
1104 */
1105 inline int32_t indexOf(UChar32 c,
1106 int32_t start) const;
1107
1108 /**
1109 * Locate in this the first occurrence of the BMP code point `c`
1110 * in the range [`start`, `start + length`),
1111 * using bitwise comparison.
1112 * @param c The code unit to search for.
1113 * @param start the offset into this at which to start matching
1114 * @param length the number of characters in this to search
1115 * @return The offset into this of `c`, or -1 if not found.
1116 * @stable ICU 2.0
1117 */
1118 inline int32_t indexOf(char16_t c,
1119 int32_t start,
1120 int32_t length) const;
1121
1122 /**
1123 * Locate in this the first occurrence of the code point `c`
1124 * in the range [`start`, `start + length`),
1125 * using bitwise comparison.
1126 *
1127 * @param c The code point to search for.
1128 * @param start the offset into this at which to start matching
1129 * @param length the number of characters in this to search
1130 * @return The offset into this of `c`, or -1 if not found.
1131 * @stable ICU 2.0
1132 */
1133 inline int32_t indexOf(UChar32 c,
1134 int32_t start,
1135 int32_t length) const;
1136
1137 /**
1138 * Locate in this the last occurrence of the characters in `text`,
1139 * using bitwise comparison.
1140 * @param text The text to search for.
1141 * @return The offset into this of the start of `text`,
1142 * or -1 if not found.
1143 * @stable ICU 2.0
1144 */
1145 inline int32_t lastIndexOf(const UnicodeString& text) const;
1146
1147 /**
1148 * Locate in this the last occurrence of the characters in `text`
1149 * starting at offset `start`, using bitwise comparison.
1150 * @param text The text to search for.
1151 * @param start The offset at which searching will start.
1152 * @return The offset into this of the start of `text`,
1153 * or -1 if not found.
1154 * @stable ICU 2.0
1155 */
1156 inline int32_t lastIndexOf(const UnicodeString& text,
1157 int32_t start) const;
1158
1159 /**
1160 * Locate in this the last occurrence in the range
1161 * [`start`, `start + length`) of the characters
1162 * in `text`, using bitwise comparison.
1163 * @param text The text to search for.
1164 * @param start The offset at which searching will start.
1165 * @param length The number of characters to search
1166 * @return The offset into this of the start of `text`,
1167 * or -1 if not found.
1168 * @stable ICU 2.0
1169 */
1170 inline int32_t lastIndexOf(const UnicodeString& text,
1171 int32_t start,
1172 int32_t length) const;
1173
1174 /**
1175 * Locate in this the last occurrence in the range
1176 * [`start`, `start + length`) of the characters
1177 * in `srcText` in the range
1178 * [`srcStart`, `srcStart + srcLength`),
1179 * using bitwise comparison.
1180 * @param srcText The text to search for.
1181 * @param srcStart the offset into `srcText` at which
1182 * to start matching
1183 * @param srcLength the number of characters in `srcText` to match
1184 * @param start the offset into this at which to start matching
1185 * @param length the number of characters in this to search
1186 * @return The offset into this of the start of `text`,
1187 * or -1 if not found.
1188 * @stable ICU 2.0
1189 */
1190 inline int32_t lastIndexOf(const UnicodeString& srcText,
1191 int32_t srcStart,
1192 int32_t srcLength,
1193 int32_t start,
1194 int32_t length) const;
1195
1196 /**
1197 * Locate in this the last occurrence of the characters in `srcChars`
1198 * starting at offset `start`, using bitwise comparison.
1199 * @param srcChars The text to search for.
1200 * @param srcLength the number of characters in `srcChars` to match
