Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes16kdownloads
uchar.h4057 linesDownload Raw Back to unicode
1// © 2016 and later: Unicode, Inc. and others.
2// License & terms of use: http://www.unicode.org/copyright.html
3/*
4**********************************************************************
5*   Copyright (C) 1997-2016, International Business Machines
6*   Corporation and others.  All Rights Reserved.
7**********************************************************************
8*
9* File UCHAR.H
10*
11* Modification History:
12*
13*   Date        Name        Description
14*   04/02/97    aliu        Creation.
15*   03/29/99    helena      Updated for C APIs.
16*   4/15/99     Madhu       Updated for C Implementation and Javadoc
17*   5/20/99     Madhu       Added the function u_getVersion()
18*   8/19/1999   srl         Upgraded scripts to Unicode 3.0
19*   8/27/1999   schererm    UCharDirection constants: U_...
20*   11/11/1999  weiv        added u_isalnum(), cleaned comments
21*   01/11/2000  helena      Renamed u_getVersion to u_getUnicodeVersion().
22******************************************************************************
23*/
24
25#ifndef UCHAR_H
26#define UCHAR_H
27
28#include "unicode/utypes.h"
29#include "unicode/stringoptions.h"
30#include "unicode/ucpmap.h"
31
32#if !defined(USET_DEFINED) && !defined(U_IN_DOXYGEN)
33
34#define USET_DEFINED
35
36/**
37 * USet is the C API type corresponding to C++ class UnicodeSet.
38 * It is forward-declared here to avoid including unicode/uset.h file if related
39 * APIs are not used.
40 *
41 * @see ucnv_getUnicodeSet
42 * @stable ICU 2.4
43 */
44typedef struct USet USet;
45
46#endif
47
48
49U_CDECL_BEGIN
50
51/*==========================================================================*/
52/* Unicode version number                                                   */
53/*==========================================================================*/
54/**
55 * Unicode version number, default for the current ICU version.
56 * The actual Unicode Character Database (UCD) data is stored in uprops.dat
57 * and may be generated from UCD files from a different Unicode version.
58 * Call u_getUnicodeVersion to get the actual Unicode version of the data.
59 *
60 * @see u_getUnicodeVersion
61 * @stable ICU 2.0
62 */
63#define U_UNICODE_VERSION "13.0"
64
65/**
66 * \file
67 * \brief C API: Unicode Properties
68 *
69 * This C API provides low-level access to the Unicode Character Database.
70 * In addition to raw property values, some convenience functions calculate
71 * derived properties, for example for Java-style programming.
72 *
73 * Unicode assigns each code point (not just assigned character) values for
74 * many properties.
75 * Most of them are simple boolean flags, or constants from a small enumerated list.
76 * For some properties, values are strings or other relatively more complex types.
77 *
78 * For more information see
79 * "About the Unicode Character Database" (http://www.unicode.org/ucd/)
80 * and the ICU User Guide chapter on Properties (http://icu-project.org/userguide/properties.html).
81 *
82 * Many properties are accessible via generic functions that take a UProperty selector.
83 * - u_hasBinaryProperty() returns a binary value (TRUE/FALSE) per property and code point.
84 * - u_getIntPropertyValue() returns an integer value per property and code point.
85 *   For each supported enumerated or catalog property, there is
86 *   an enum type for all of the property's values, and
87 *   u_getIntPropertyValue() returns the numeric values of those constants.
88 * - u_getBinaryPropertySet() returns a set for each ICU-supported binary property with
89 *   all code points for which the property is true.
90 * - u_getIntPropertyMap() returns a map for each
91 *   ICU-supported enumerated/catalog/int-valued property which
92 *   maps all Unicode code points to their values for that property.
93 *
94 * Many functions are designed to match java.lang.Character functions.
95 * See the individual function documentation,
96 * and see the JDK 1.4 java.lang.Character documentation
97 * at http://java.sun.com/j2se/1.4/docs/api/java/lang/Character.html
98 *
99 * There are also functions that provide easy migration from C/POSIX functions
100 * like isblank(). Their use is generally discouraged because the C/POSIX
101 * standards do not define their semantics beyond the ASCII range, which means
102 * that different implementations exhibit very different behavior.
103 * Instead, Unicode properties should be used directly.
104 *
105 * There are also only a few, broad C/POSIX character classes, and they tend
106 * to be used for conflicting purposes. For example, the "isalpha()" class
107 * is sometimes used to determine word boundaries, while a more sophisticated
108 * approach would at least distinguish initial letters from continuation
109 * characters (the latter including combining marks).
110 * (In ICU, BreakIterator is the most sophisticated API for word boundaries.)
111 * Another example: There is no "istitle()" class for titlecase characters.
112 *
113 * ICU 3.4 and later provides API access for all twelve C/POSIX character classes.
114 * ICU implements them according to the Standard Recommendations in
115 * Annex C: Compatibility Properties of UTS #18 Unicode Regular Expressions
116 * (http://www.unicode.org/reports/tr18/#Compatibility_Properties).
117 *
118 * API access for C/POSIX character classes is as follows:
119 * - alpha:     u_isUAlphabetic(c) or u_hasBinaryProperty(c, UCHAR_ALPHABETIC)
120 * - lower:     u_isULowercase(c) or u_hasBinaryProperty(c, UCHAR_LOWERCASE)
121 * - upper:     u_isUUppercase(c) or u_hasBinaryProperty(c, UCHAR_UPPERCASE)
122 * - punct:     u_ispunct(c)
123 * - digit:     u_isdigit(c) or u_charType(c)==U_DECIMAL_DIGIT_NUMBER
124 * - xdigit:    u_isxdigit(c) or u_hasBinaryProperty(c, UCHAR_POSIX_XDIGIT)
125 * - alnum:     u_hasBinaryProperty(c, UCHAR_POSIX_ALNUM)
126 * - space:     u_isUWhiteSpace(c) or u_hasBinaryProperty(c, UCHAR_WHITE_SPACE)
127 * - blank:     u_isblank(c) or u_hasBinaryProperty(c, UCHAR_POSIX_BLANK)
128 * - cntrl:     u_charType(c)==U_CONTROL_CHAR
129 * - graph:     u_hasBinaryProperty(c, UCHAR_POSIX_GRAPH)
130 * - print:     u_hasBinaryProperty(c, UCHAR_POSIX_PRINT)
131 *
132 * Note: Some of the u_isxyz() functions in uchar.h predate, and do not match,
133 * the Standard Recommendations in UTS #18. Instead, they match Java
134 * functions according to their API documentation.
135 *
136 * \htmlonly
137 * The C/POSIX character classes are also available in UnicodeSet patterns,
138 * using patterns like [:graph:] or \p{graph}.
139 * \endhtmlonly
140 *
141 * Note: There are several ICU whitespace functions.
142 * Comparison:
143 * - u_isUWhiteSpace=UCHAR_WHITE_SPACE: Unicode White_Space property;
144 *       most of general categories "Z" (separators) + most whitespace ISO controls
145 *       (including no-break spaces, but excluding IS1..IS4)
146 * - u_isWhitespace: Java isWhitespace; Z + whitespace ISO controls but excluding no-break spaces
147 * - u_isJavaSpaceChar: Java isSpaceChar; just Z (including no-break spaces)
148 * - u_isspace: Z + whitespace ISO controls (including no-break spaces)
149 * - u_isblank: "horizontal spaces" = TAB + Zs
150 */
151
152/**
153 * Constants.
154 */
155
156/** The lowest Unicode code point value. Code points are non-negative. @stable ICU 2.0 */
157#define UCHAR_MIN_VALUE 0
158
159/**
160 * The highest Unicode code point value (scalar value) according to
161 * The Unicode Standard. This is a 21-bit value (20.1 bits, rounded up).
162 * For a single character, UChar32 is a simple type that can hold any code point value.
163 *
164 * @see UChar32
165 * @stable ICU 2.0
166 */
167#define UCHAR_MAX_VALUE 0x10ffff
168
169/**
170 * Get a single-bit bit set (a flag) from a bit number 0..31.
171 * @stable ICU 2.1
172 */
173#define U_MASK(x) ((uint32_t)1<<(x))
174
175/**
176 * Selection constants for Unicode properties.
177 * These constants are used in functions like u_hasBinaryProperty to select
178 * one of the Unicode properties.
179 *
180 * The properties APIs are intended to reflect Unicode properties as defined
181 * in the Unicode Character Database (UCD) and Unicode Technical Reports (UTR).
182 *
183 * For details about the properties see
184 * UAX #44: Unicode Character Database (http://www.unicode.org/reports/tr44/).
185 *
186 * Important: If ICU is built with UCD files from Unicode versions below, e.g., 3.2,
187 * then properties marked with "new in Unicode 3.2" are not or not fully available.
188 * Check u_getUnicodeVersion to be sure.
189 *
190 * @see u_hasBinaryProperty
191 * @see u_getIntPropertyValue
192 * @see u_getUnicodeVersion
193 * @stable ICU 2.1
194 */
195typedef enum UProperty {
196    /*
197     * Note: UProperty constants are parsed by preparseucd.py.
198     * It matches lines like
199     *     UCHAR_<Unicode property name>=<integer>,
200     */
201
202    /*  Note: Place UCHAR_ALPHABETIC before UCHAR_BINARY_START so that
203    debuggers display UCHAR_ALPHABETIC as the symbolic name for 0,
204    rather than UCHAR_BINARY_START.  Likewise for other *_START
205    identifiers. */
206
207    /** Binary property Alphabetic. Same as u_isUAlphabetic, different from u_isalpha.
208        Lu+Ll+Lt+Lm+Lo+Nl+Other_Alphabetic @stable ICU 2.1 */
209    UCHAR_ALPHABETIC=0,
210    /** First constant for binary Unicode properties. @stable ICU 2.1 */
211    UCHAR_BINARY_START=UCHAR_ALPHABETIC,
212    /** Binary property ASCII_Hex_Digit. 0-9 A-F a-f @stable ICU 2.1 */
213    UCHAR_ASCII_HEX_DIGIT=1,
214    /** Binary property Bidi_Control.
215        Format controls which have specific functions
216        in the Bidi Algorithm. @stable ICU 2.1 */
217    UCHAR_BIDI_CONTROL=2,
218    /** Binary property Bidi_Mirrored.
219        Characters that may change display in RTL text.
220        Same as u_isMirrored.
221        See Bidi Algorithm, UTR 9. @stable ICU 2.1 */
222    UCHAR_BIDI_MIRRORED=3,
223    /** Binary property Dash. Variations of dashes. @stable ICU 2.1 */
224    UCHAR_DASH=4,
225    /** Binary property Default_Ignorable_Code_Point (new in Unicode 3.2).
226        Ignorable in most processing.
227        <2060..206F, FFF0..FFFB, E0000..E0FFF>+Other_Default_Ignorable_Code_Point+(Cf+Cc+Cs-White_Space) @stable ICU 2.1 */
228    UCHAR_DEFAULT_IGNORABLE_CODE_POINT=5,
229    /** Binary property Deprecated (new in Unicode 3.2).
230        The usage of deprecated characters is strongly discouraged. @stable ICU 2.1 */
231    UCHAR_DEPRECATED=6,
232    /** Binary property Diacritic. Characters that linguistically modify
233        the meaning of another character to which they apply. @stable ICU 2.1 */
234    UCHAR_DIACRITIC=7,
235    /** Binary property Extender.
236        Extend the value or shape of a preceding alphabetic character,
237        e.g., length and iteration marks. @stable ICU 2.1 */
238    UCHAR_EXTENDER=8,
239    /** Binary property Full_Composition_Exclusion.
240        CompositionExclusions.txt+Singleton Decompositions+
241        Non-Starter Decompositions. @stable ICU 2.1 */
242    UCHAR_FULL_COMPOSITION_EXCLUSION=9,
243    /** Binary property Grapheme_Base (new in Unicode 3.2).
244        For programmatic determination of grapheme cluster boundaries.
245        [0..10FFFF]-Cc-Cf-Cs-Co-Cn-Zl-Zp-Grapheme_Link-Grapheme_Extend-CGJ @stable ICU 2.1 */
246    UCHAR_GRAPHEME_BASE=10,
247    /** Binary property Grapheme_Extend (new in Unicode 3.2).
248        For programmatic determination of grapheme cluster boundaries.
249        Me+Mn+Mc+Other_Grapheme_Extend-Grapheme_Link-CGJ @stable ICU 2.1 */
250    UCHAR_GRAPHEME_EXTEND=11,
251    /** Binary property Grapheme_Link (new in Unicode 3.2).
252        For programmatic determination of grapheme cluster boundaries. @stable ICU 2.1 */
253    UCHAR_GRAPHEME_LINK=12,
254    /** Binary property Hex_Digit.
255        Characters commonly used for hexadecimal numbers. @stable ICU 2.1 */
256    UCHAR_HEX_DIGIT=13,
257    /** Binary property Hyphen. Dashes used to mark connections
258        between pieces of words, plus the Katakana middle dot. @stable ICU 2.1 */
259    UCHAR_HYPHEN=14,
260    /** Binary property ID_Continue.
261        Characters that can continue an identifier.
262        DerivedCoreProperties.txt also says "NOTE: Cf characters should be filtered out."
263        ID_Start+Mn+Mc+Nd+Pc @stable ICU 2.1 */
264    UCHAR_ID_CONTINUE=15,
265    /** Binary property ID_Start.
266        Characters that can start an identifier.
267        Lu+Ll+Lt+Lm+Lo+Nl @stable ICU 2.1 */
268    UCHAR_ID_START=16,
269    /** Binary property Ideographic.
270        CJKV ideographs. @stable ICU 2.1 */
271    UCHAR_IDEOGRAPHIC=17,
272    /** Binary property IDS_Binary_Operator (new in Unicode 3.2).
273        For programmatic determination of
274        Ideographic Description Sequences. @stable ICU 2.1 */
275    UCHAR_IDS_BINARY_OPERATOR=18,
276    /** Binary property IDS_Trinary_Operator (new in Unicode 3.2).
277        For programmatic determination of
278        Ideographic Description Sequences. @stable ICU 2.1 */
279    UCHAR_IDS_TRINARY_OPERATOR=19,
280    /** Binary property Join_Control.
281        Format controls for cursive joining and ligation. @stable ICU 2.1 */
282    UCHAR_JOIN_CONTROL=20,
283    /** Binary property Logical_Order_Exception (new in Unicode 3.2).
284        Characters that do not use logical order and
285        require special handling in most processing. @stable ICU 2.1 */
286    UCHAR_LOGICAL_ORDER_EXCEPTION=21,
287    /** Binary property Lowercase. Same as u_isULowercase, different from u_islower.
288        Ll+Other_Lowercase @stable ICU 2.1 */
289    UCHAR_LOWERCASE=22,
290    /** Binary property Math. Sm+Other_Math @stable ICU 2.1 */
291    UCHAR_MATH=23,
292    /** Binary property Noncharacter_Code_Point.
293        Code points that are explicitly defined as illegal
294        for the encoding of characters. @stable ICU 2.1 */
295    UCHAR_NONCHARACTER_CODE_POINT=24,
296    /** Binary property Quotation_Mark. @stable ICU 2.1 */
297    UCHAR_QUOTATION_MARK=25,
298    /** Binary property Radical (new in Unicode 3.2).
299        For programmatic determination of
300        Ideographic Description Sequences. @stable ICU 2.1 */
301    UCHAR_RADICAL=26,
302    /** Binary property Soft_Dotted (new in Unicode 3.2).
303        Characters with a "soft dot", like i or j.
304        An accent placed on these characters causes
305        the dot to disappear. @stable ICU 2.1 */
306    UCHAR_SOFT_DOTTED=27,
307    /** Binary property Terminal_Punctuation.
308        Punctuation characters that generally mark
309        the end of textual units. @stable ICU 2.1 */
310    UCHAR_TERMINAL_PUNCTUATION=28,
311    /** Binary property Unified_Ideograph (new in Unicode 3.2).
312        For programmatic determination of
313        Ideographic Description Sequences. @stable ICU 2.1 */
314    UCHAR_UNIFIED_IDEOGRAPH=29,
315    /** Binary property Uppercase. Same as u_isUUppercase, different from u_isupper.
316        Lu+Other_Uppercase @stable ICU 2.1 */
317    UCHAR_UPPERCASE=30,
318    /** Binary property White_Space.
319        Same as u_isUWhiteSpace, different from u_isspace and u_isWhitespace.
320        Space characters+TAB+CR+LF-ZWSP-ZWNBSP @stable ICU 2.1 */
321    UCHAR_WHITE_SPACE=31,
322    /** Binary property XID_Continue.
323        ID_Continue modified to allow closure under
324        normalization forms NFKC and NFKD. @stable ICU 2.1 */
325    UCHAR_XID_CONTINUE=32,
326    /** Binary property XID_Start. ID_Start modified to allow
327        closure under normalization forms NFKC and NFKD. @stable ICU 2.1 */
328    UCHAR_XID_START=33,
329    /** Binary property Case_Sensitive. Either the source of a case
330        mapping or _in_ the target of a case mapping. Not the same as
331        the general category Cased_Letter. @stable ICU 2.6 */
332   UCHAR_CASE_SENSITIVE=34,
333    /** Binary property STerm (new in Unicode 4.0.1).
334        Sentence Terminal. Used in UAX #29: Text Boundaries
335        (http://www.unicode.org/reports/tr29/)
336        @stable ICU 3.0 */
337    UCHAR_S_TERM=35,
338    /** Binary property Variation_Selector (new in Unicode 4.0.1).
339        Indicates all those characters that qualify as Variation Selectors.
340        For details on the behavior of these characters,
341        see StandardizedVariants.html and 15.6 Variation Selectors.
342        @stable ICU 3.0 */
343    UCHAR_VARIATION_SELECTOR=36,
344    /** Binary property NFD_Inert.
345        ICU-specific property for characters that are inert under NFD,
346        i.e., they do not interact with adjacent characters.
347        See the documentation for the Normalizer2 class and the
348        Normalizer2::isInert() method.
349        @stable ICU 3.0 */
350    UCHAR_NFD_INERT=37,
351    /** Binary property NFKD_Inert.
352        ICU-specific property for characters that are inert under NFKD,
353        i.e., they do not interact with adjacent characters.
354        See the documentation for the Normalizer2 class and the
355        Normalizer2::isInert() method.
356        @stable ICU 3.0 */
357    UCHAR_NFKD_INERT=38,
358    /** Binary property NFC_Inert.
359        ICU-specific property for characters that are inert under NFC,
360        i.e., they do not interact with adjacent characters.
361        See the documentation for the Normalizer2 class and the
362        Normalizer2::isInert() method.
363        @stable ICU 3.0 */
364    UCHAR_NFC_INERT=39,
365    /** Binary property NFKC_Inert.
366        ICU-specific property for characters that are inert under NFKC,
367        i.e., they do not interact with adjacent characters.
368        See the documentation for the Normalizer2 class and the
369        Normalizer2::isInert() method.
370        @stable ICU 3.0 */
371    UCHAR_NFKC_INERT=40,
372    /** Binary Property Segment_Starter.
373        ICU-specific property for characters that are starters in terms of
374        Unicode normalization and combining character sequences.
375        They have ccc=0 and do not occur in non-initial position of the
376        canonical decomposition of any character
377        (like a-umlaut in NFD and a Jamo T in an NFD(Hangul LVT)).
378        ICU uses this property for segmenting a string for generating a set of
379        canonically equivalent strings, e.g. for canonical closure while
380        processing collation tailoring rules.
381        @stable ICU 3.0 */
382    UCHAR_SEGMENT_STARTER=41,
383    /** Binary property Pattern_Syntax (new in Unicode 4.1).
384        See UAX #31 Identifier and Pattern Syntax
385        (http://www.unicode.org/reports/tr31/)
386        @stable ICU 3.4 */
387    UCHAR_PATTERN_SYNTAX=42,
388    /** Binary property Pattern_White_Space (new in Unicode 4.1).
389        See UAX #31 Identifier and Pattern Syntax
390        (http://www.unicode.org/reports/tr31/)
391        @stable ICU 3.4 */
392    UCHAR_PATTERN_WHITE_SPACE=43,
393    /** Binary property alnum (a C/POSIX character class).
394        Implemented according to the UTS #18 Annex C Standard Recommendation.
395        See the uchar.h file documentation.
396        @stable ICU 3.4 */
397    UCHAR_POSIX_ALNUM=44,
398    /** Binary property blank (a C/POSIX character class).
399        Implemented according to the UTS #18 Annex C Standard Recommendation.
400        See the uchar.h file documentation.
401        @stable ICU 3.4 */
402    UCHAR_POSIX_BLANK=45,
403    /** Binary property graph (a C/POSIX character class).
404        Implemented according to the UTS #18 Annex C Standard Recommendation.
405        See the uchar.h file documentation.
406        @stable ICU 3.4 */
407    UCHAR_POSIX_GRAPH=46,
408    /** Binary property print (a C/POSIX character class).
409        Implemented according to the UTS #18 Annex C Standard Recommendation.
410        See the uchar.h file documentation.
411        @stable ICU 3.4 */
412    UCHAR_POSIX_PRINT=47,
413    /** Binary property xdigit (a C/POSIX character class).
414        Implemented according to the UTS #18 Annex C Standard Recommendation.
415        See the uchar.h file documentation.
416        @stable ICU 3.4 */
417    UCHAR_POSIX_XDIGIT=48,
418    /** Binary property Cased. For Lowercase, Uppercase and Titlecase characters. @stable ICU 4.4 */
419    UCHAR_CASED=49,
420    /** Binary property Case_Ignorable. Used in context-sensitive case mappings. @stable ICU 4.4 */
421    UCHAR_CASE_IGNORABLE=50,
422    /** Binary property Changes_When_Lowercased. @stable ICU 4.4 */
423    UCHAR_CHANGES_WHEN_LOWERCASED=51,
424    /** Binary property Changes_When_Uppercased. @stable ICU 4.4 */
425    UCHAR_CHANGES_WHEN_UPPERCASED=52,
426    /** Binary property Changes_When_Titlecased. @stable ICU 4.4 */
427    UCHAR_CHANGES_WHEN_TITLECASED=53,
428    /** Binary property Changes_When_Casefolded. @stable ICU 4.4 */
429    UCHAR_CHANGES_WHEN_CASEFOLDED=54,
430    /** Binary property Changes_When_Casemapped. @stable ICU 4.4 */
431    UCHAR_CHANGES_WHEN_CASEMAPPED=55,
432    /** Binary property Changes_When_NFKC_Casefolded. @stable ICU 4.4 */
433    UCHAR_CHANGES_WHEN_NFKC_CASEFOLDED=56,
434    /**
435     * Binary property Emoji.
436     * See http://www.unicode.org/reports/tr51/#Emoji_Properties
437     *
438     * @stable ICU 57
439     */
440    UCHAR_EMOJI=57,
441    /**
442     * Binary property Emoji_Presentation.
443     * See http://www.unicode.org/reports/tr51/#Emoji_Properties
444     *
445     * @stable ICU 57
446     */
447    UCHAR_EMOJI_PRESENTATION=58,
448    /**
449     * Binary property Emoji_Modifier.
450     * See http://www.unicode.org/reports/tr51/#Emoji_Properties
451     *
452     * @stable ICU 57
453     */
454    UCHAR_EMOJI_MODIFIER=59,
455    /**
456     * Binary property Emoji_Modifier_Base.
457     * See http://www.unicode.org/reports/tr51/#Emoji_Properties
458     *
459     * @stable ICU 57
460     */
461    UCHAR_EMOJI_MODIFIER_BASE=60,
462    /**
463     * Binary property Emoji_Component.
464     * See http://www.unicode.org/reports/tr51/#Emoji_Properties
465     *
466     * @stable ICU 60
467     */
468    UCHAR_EMOJI_COMPONENT=61,
469    /**
470     * Binary property Regional_Indicator.
471     * @stable ICU 60
472     */
473    UCHAR_REGIONAL_INDICATOR=62,
474    /**
475     * Binary property Prepended_Concatenation_Mark.
476     * @stable ICU 60
477     */
478    UCHAR_PREPENDED_CONCATENATION_MARK=63,
479    /**
480     * Binary property Extended_Pictographic.
481     * See http://www.unicode.org/reports/tr51/#Emoji_Properties
482     *
483     * @stable ICU 62
484     */
485    UCHAR_EXTENDED_PICTOGRAPHIC=64,
486#ifndef U_HIDE_DEPRECATED_API
487    /**
488     * One more than the last constant for binary Unicode properties.
489     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
490     */
491    UCHAR_BINARY_LIMIT,
492#endif  // U_HIDE_DEPRECATED_API
493
494    /** Enumerated property Bidi_Class.
495        Same as u_charDirection, returns UCharDirection values. @stable ICU 2.2 */
496    UCHAR_BIDI_CLASS=0x1000,
497    /** First constant for enumerated/integer Unicode properties. @stable ICU 2.2 */
498    UCHAR_INT_START=UCHAR_BIDI_CLASS,
499    /** Enumerated property Block.
500        Same as ublock_getCode, returns UBlockCode values. @stable ICU 2.2 */
501    UCHAR_BLOCK=0x1001,
502    /** Enumerated property Canonical_Combining_Class.
503        Same as u_getCombiningClass, returns 8-bit numeric values. @stable ICU 2.2 */
504    UCHAR_CANONICAL_COMBINING_CLASS=0x1002,
505    /** Enumerated property Decomposition_Type.
506        Returns UDecompositionType values. @stable ICU 2.2 */
507    UCHAR_DECOMPOSITION_TYPE=0x1003,
508    /** Enumerated property East_Asian_Width.
509        See http://www.unicode.org/reports/tr11/
510        Returns UEastAsianWidth values. @stable ICU 2.2 */
511    UCHAR_EAST_ASIAN_WIDTH=0x1004,
512    /** Enumerated property General_Category.
513        Same as u_charType, returns UCharCategory values. @stable ICU 2.2 */
514    UCHAR_GENERAL_CATEGORY=0x1005,
515    /** Enumerated property Joining_Group.
516        Returns UJoiningGroup values. @stable ICU 2.2 */
517    UCHAR_JOINING_GROUP=0x1006,
518    /** Enumerated property Joining_Type.
519        Returns UJoiningType values. @stable ICU 2.2 */
520    UCHAR_JOINING_TYPE=0x1007,
521    /** Enumerated property Line_Break.
522        Returns ULineBreak values. @stable ICU 2.2 */
523    UCHAR_LINE_BREAK=0x1008,
524    /** Enumerated property Numeric_Type.
525        Returns UNumericType values. @stable ICU 2.2 */
526    UCHAR_NUMERIC_TYPE=0x1009,
527    /** Enumerated property Script.
528        Same as uscript_getScript, returns UScriptCode values. @stable ICU 2.2 */
529    UCHAR_SCRIPT=0x100A,
530    /** Enumerated property Hangul_Syllable_Type, new in Unicode 4.
531        Returns UHangulSyllableType values. @stable ICU 2.6 */
532    UCHAR_HANGUL_SYLLABLE_TYPE=0x100B,
533    /** Enumerated property NFD_Quick_Check.
534        Returns UNormalizationCheckResult values. @stable ICU 3.0 */
535    UCHAR_NFD_QUICK_CHECK=0x100C,
536    /** Enumerated property NFKD_Quick_Check.
537        Returns UNormalizationCheckResult values. @stable ICU 3.0 */
538    UCHAR_NFKD_QUICK_CHECK=0x100D,
539    /** Enumerated property NFC_Quick_Check.
540        Returns UNormalizationCheckResult values. @stable ICU 3.0 */
541    UCHAR_NFC_QUICK_CHECK=0x100E,
542    /** Enumerated property NFKC_Quick_Check.
543        Returns UNormalizationCheckResult values. @stable ICU 3.0 */
544    UCHAR_NFKC_QUICK_CHECK=0x100F,
545    /** Enumerated property Lead_Canonical_Combining_Class.
546        ICU-specific property for the ccc of the first code point
547        of the decomposition, or lccc(c)=ccc(NFD(c)[0]).
548        Useful for checking for canonically ordered text;
549        see UNORM_FCD and http://www.unicode.org/notes/tn5/#FCD .
550        Returns 8-bit numeric values like UCHAR_CANONICAL_COMBINING_CLASS. @stable ICU 3.0 */
551    UCHAR_LEAD_CANONICAL_COMBINING_CLASS=0x1010,
552    /** Enumerated property Trail_Canonical_Combining_Class.
553        ICU-specific property for the ccc of the last code point
554        of the decomposition, or tccc(c)=ccc(NFD(c)[last]).
555        Useful for checking for canonically ordered text;
556        see UNORM_FCD and http://www.unicode.org/notes/tn5/#FCD .
557        Returns 8-bit numeric values like UCHAR_CANONICAL_COMBINING_CLASS. @stable ICU 3.0 */
558    UCHAR_TRAIL_CANONICAL_COMBINING_CLASS=0x1011,
559    /** Enumerated property Grapheme_Cluster_Break (new in Unicode 4.1).
560        Used in UAX #29: Text Boundaries
561        (http://www.unicode.org/reports/tr29/)
562        Returns UGraphemeClusterBreak values. @stable ICU 3.4 */
563    UCHAR_GRAPHEME_CLUSTER_BREAK=0x1012,
564    /** Enumerated property Sentence_Break (new in Unicode 4.1).
565        Used in UAX #29: Text Boundaries
566        (http://www.unicode.org/reports/tr29/)
567        Returns USentenceBreak values. @stable ICU 3.4 */
568    UCHAR_SENTENCE_BREAK=0x1013,
569    /** Enumerated property Word_Break (new in Unicode 4.1).
570        Used in UAX #29: Text Boundaries
571        (http://www.unicode.org/reports/tr29/)
572        Returns UWordBreakValues values. @stable ICU 3.4 */
573    UCHAR_WORD_BREAK=0x1014,
574    /** Enumerated property Bidi_Paired_Bracket_Type (new in Unicode 6.3).
575        Used in UAX #9: Unicode Bidirectional Algorithm
576        (http://www.unicode.org/reports/tr9/)
577        Returns UBidiPairedBracketType values. @stable ICU 52 */
578    UCHAR_BIDI_PAIRED_BRACKET_TYPE=0x1015,
579    /**
580     * Enumerated property Indic_Positional_Category.
581     * New in Unicode 6.0 as provisional property Indic_Matra_Category;
582     * renamed and changed to informative in Unicode 8.0.
583     * See http://www.unicode.org/reports/tr44/#IndicPositionalCategory.txt
584     * @stable ICU 63
585     */
586    UCHAR_INDIC_POSITIONAL_CATEGORY=0x1016,
587    /**
588     * Enumerated property Indic_Syllabic_Category.
589     * New in Unicode 6.0 as provisional; informative since Unicode 8.0.
590     * See http://www.unicode.org/reports/tr44/#IndicSyllabicCategory.txt
591     * @stable ICU 63
592     */
593    UCHAR_INDIC_SYLLABIC_CATEGORY=0x1017,
594    /**
595     * Enumerated property Vertical_Orientation.
596     * Used for UAX #50 Unicode Vertical Text Layout (https://www.unicode.org/reports/tr50/).
597     * New as a UCD property in Unicode 10.0.
598     * @stable ICU 63
599     */
600    UCHAR_VERTICAL_ORIENTATION=0x1018,
601#ifndef U_HIDE_DEPRECATED_API
602    /**
603     * One more than the last constant for enumerated/integer Unicode properties.
604     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
605     */
606    UCHAR_INT_LIMIT=0x1019,
607#endif  // U_HIDE_DEPRECATED_API
608
609    /** Bitmask property General_Category_Mask.
610        This is the General_Category property returned as a bit mask.
611        When used in u_getIntPropertyValue(c), same as U_MASK(u_charType(c)),
612        returns bit masks for UCharCategory values where exactly one bit is set.
613        When used with u_getPropertyValueName() and u_getPropertyValueEnum(),
614        a multi-bit mask is used for sets of categories like "Letters".
615        Mask values should be cast to uint32_t.
616        @stable ICU 2.4 */
617    UCHAR_GENERAL_CATEGORY_MASK=0x2000,
618    /** First constant for bit-mask Unicode properties. @stable ICU 2.4 */
619    UCHAR_MASK_START=UCHAR_GENERAL_CATEGORY_MASK,
620#ifndef U_HIDE_DEPRECATED_API
621    /**
622     * One more than the last constant for bit-mask Unicode properties.
623     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
624     */
625    UCHAR_MASK_LIMIT=0x2001,
626#endif  // U_HIDE_DEPRECATED_API
627
628    /** Double property Numeric_Value.
629        Corresponds to u_getNumericValue. @stable ICU 2.4 */
630    UCHAR_NUMERIC_VALUE=0x3000,
631    /** First constant for double Unicode properties. @stable ICU 2.4 */
632    UCHAR_DOUBLE_START=UCHAR_NUMERIC_VALUE,
633#ifndef U_HIDE_DEPRECATED_API
634    /**
635     * One more than the last constant for double Unicode properties.
636     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
637     */
638    UCHAR_DOUBLE_LIMIT=0x3001,
639#endif  // U_HIDE_DEPRECATED_API
640
641    /** String property Age.
642        Corresponds to u_charAge. @stable ICU 2.4 */
643    UCHAR_AGE=0x4000,
644    /** First constant for string Unicode properties. @stable ICU 2.4 */
645    UCHAR_STRING_START=UCHAR_AGE,
646    /** String property Bidi_Mirroring_Glyph.
647        Corresponds to u_charMirror. @stable ICU 2.4 */
648    UCHAR_BIDI_MIRRORING_GLYPH=0x4001,
649    /** String property Case_Folding.
650        Corresponds to u_strFoldCase in ustring.h. @stable ICU 2.4 */
651    UCHAR_CASE_FOLDING=0x4002,
652#ifndef U_HIDE_DEPRECATED_API
653    /** Deprecated string property ISO_Comment.
654        Corresponds to u_getISOComment. @deprecated ICU 49 */
655    UCHAR_ISO_COMMENT=0x4003,
656#endif  /* U_HIDE_DEPRECATED_API */
657    /** String property Lowercase_Mapping.
658        Corresponds to u_strToLower in ustring.h. @stable ICU 2.4 */
659    UCHAR_LOWERCASE_MAPPING=0x4004,
660    /** String property Name.
661        Corresponds to u_charName. @stable ICU 2.4 */
662    UCHAR_NAME=0x4005,
663    /** String property Simple_Case_Folding.
664        Corresponds to u_foldCase. @stable ICU 2.4 */
665    UCHAR_SIMPLE_CASE_FOLDING=0x4006,
666    /** String property Simple_Lowercase_Mapping.
667        Corresponds to u_tolower. @stable ICU 2.4 */
668    UCHAR_SIMPLE_LOWERCASE_MAPPING=0x4007,
669    /** String property Simple_Titlecase_Mapping.
670        Corresponds to u_totitle. @stable ICU 2.4 */
671    UCHAR_SIMPLE_TITLECASE_MAPPING=0x4008,
672    /** String property Simple_Uppercase_Mapping.
673        Corresponds to u_toupper. @stable ICU 2.4 */
674    UCHAR_SIMPLE_UPPERCASE_MAPPING=0x4009,
675    /** String property Titlecase_Mapping.
676        Corresponds to u_strToTitle in ustring.h. @stable ICU 2.4 */
677    UCHAR_TITLECASE_MAPPING=0x400A,
678#ifndef U_HIDE_DEPRECATED_API
679    /** String property Unicode_1_Name.
680        This property is of little practical value.
681        Beginning with ICU 49, ICU APIs return an empty string for this property.
682        Corresponds to u_charName(U_UNICODE_10_CHAR_NAME). @deprecated ICU 49 */
683    UCHAR_UNICODE_1_NAME=0x400B,
684#endif  /* U_HIDE_DEPRECATED_API */
685    /** String property Uppercase_Mapping.
686        Corresponds to u_strToUpper in ustring.h. @stable ICU 2.4 */
687    UCHAR_UPPERCASE_MAPPING=0x400C,
688    /** String property Bidi_Paired_Bracket (new in Unicode 6.3).
689        Corresponds to u_getBidiPairedBracket. @stable ICU 52 */
690    UCHAR_BIDI_PAIRED_BRACKET=0x400D,
691#ifndef U_HIDE_DEPRECATED_API
692    /**
693     * One more than the last constant for string Unicode properties.
694     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
695     */
696    UCHAR_STRING_LIMIT=0x400E,
697#endif  // U_HIDE_DEPRECATED_API
698
699    /** Miscellaneous property Script_Extensions (new in Unicode 6.0).
700        Some characters are commonly used in multiple scripts.
701        For more information, see UAX #24: http://www.unicode.org/reports/tr24/.
702        Corresponds to uscript_hasScript and uscript_getScriptExtensions in uscript.h.
703        @stable ICU 4.6 */
704    UCHAR_SCRIPT_EXTENSIONS=0x7000,
705    /** First constant for Unicode properties with unusual value types. @stable ICU 4.6 */
706    UCHAR_OTHER_PROPERTY_START=UCHAR_SCRIPT_EXTENSIONS,
707#ifndef U_HIDE_DEPRECATED_API
708    /**
709     * One more than the last constant for Unicode properties with unusual value types.
710     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
711     */
712    UCHAR_OTHER_PROPERTY_LIMIT=0x7001,
713#endif  // U_HIDE_DEPRECATED_API
714
715    /** Represents a nonexistent or invalid property or property value. @stable ICU 2.4 */
716    UCHAR_INVALID_CODE = -1
717} UProperty;
718
719/**
720 * Data for enumerated Unicode general category types.
721 * See http://www.unicode.org/Public/UNIDATA/UnicodeData.html .
722 * @stable ICU 2.0
723 */
724typedef enum UCharCategory
725{
726    /*
727     * Note: UCharCategory constants and their API comments are parsed by preparseucd.py.
728     * It matches pairs of lines like
729     *     / ** <Unicode 2-letter General_Category value> comment... * /
730     *     U_<[A-Z_]+> = <integer>,
731     */
732
733    /** Non-category for unassigned and non-character code points. @stable ICU 2.0 */
734    U_UNASSIGNED              = 0,
735    /** Cn "Other, Not Assigned (no characters in [UnicodeData.txt] have this property)" (same as U_UNASSIGNED!) @stable ICU 2.0 */
736    U_GENERAL_OTHER_TYPES     = 0,
737    /** Lu @stable ICU 2.0 */
738    U_UPPERCASE_LETTER        = 1,
739    /** Ll @stable ICU 2.0 */
740    U_LOWERCASE_LETTER        = 2,
741    /** Lt @stable ICU 2.0 */
742    U_TITLECASE_LETTER        = 3,
743    /** Lm @stable ICU 2.0 */
744    U_MODIFIER_LETTER         = 4,
745    /** Lo @stable ICU 2.0 */
746    U_OTHER_LETTER            = 5,
747    /** Mn @stable ICU 2.0 */
748    U_NON_SPACING_MARK        = 6,
749    /** Me @stable ICU 2.0 */
750    U_ENCLOSING_MARK          = 7,
751    /** Mc @stable ICU 2.0 */
752    U_COMBINING_SPACING_MARK  = 8,
753    /** Nd @stable ICU 2.0 */
754    U_DECIMAL_DIGIT_NUMBER    = 9,
755    /** Nl @stable ICU 2.0 */
756    U_LETTER_NUMBER           = 10,
757    /** No @stable ICU 2.0 */
758    U_OTHER_NUMBER            = 11,
759    /** Zs @stable ICU 2.0 */
760    U_SPACE_SEPARATOR         = 12,
761    /** Zl @stable ICU 2.0 */
762    U_LINE_SEPARATOR          = 13,
763    /** Zp @stable ICU 2.0 */
764    U_PARAGRAPH_SEPARATOR     = 14,
765    /** Cc @stable ICU 2.0 */
766    U_CONTROL_CHAR            = 15,
767    /** Cf @stable ICU 2.0 */
768    U_FORMAT_CHAR             = 16,
769    /** Co @stable ICU 2.0 */
770    U_PRIVATE_USE_CHAR        = 17,
771    /** Cs @stable ICU 2.0 */
772    U_SURROGATE               = 18,
773    /** Pd @stable ICU 2.0 */
774    U_DASH_PUNCTUATION        = 19,
775    /** Ps @stable ICU 2.0 */
776    U_START_PUNCTUATION       = 20,
777    /** Pe @stable ICU 2.0 */
778    U_END_PUNCTUATION         = 21,
779    /** Pc @stable ICU 2.0 */
780    U_CONNECTOR_PUNCTUATION   = 22,
781    /** Po @stable ICU 2.0 */
782    U_OTHER_PUNCTUATION       = 23,
783    /** Sm @stable ICU 2.0 */
784    U_MATH_SYMBOL             = 24,
785    /** Sc @stable ICU 2.0 */
786    U_CURRENCY_SYMBOL         = 25,
787    /** Sk @stable ICU 2.0 */
788    U_MODIFIER_SYMBOL         = 26,
789    /** So @stable ICU 2.0 */
790    U_OTHER_SYMBOL            = 27,
791    /** Pi @stable ICU 2.0 */
792    U_INITIAL_PUNCTUATION     = 28,
793    /** Pf @stable ICU 2.0 */
794    U_FINAL_PUNCTUATION       = 29,
795    /**
796     * One higher than the last enum UCharCategory constant.
797     * This numeric value is stable (will not change), see
798     * http://www.unicode.org/policies/stability_policy.html#Property_Value
799     *
800     * @stable ICU 2.0
801     */
802    U_CHAR_CATEGORY_COUNT
803} UCharCategory;
804
805/**
806 * U_GC_XX_MASK constants are bit flags corresponding to Unicode
807 * general category values.
808 * For each category, the nth bit is set if the numeric value of the
809 * corresponding UCharCategory constant is n.
810 *
811 * There are also some U_GC_Y_MASK constants for groups of general categories
812 * like L for all letter categories.
813 *
814 * @see u_charType
815 * @see U_GET_GC_MASK
816 * @see UCharCategory
817 * @stable ICU 2.1
818 */
819#define U_GC_CN_MASK    U_MASK(U_GENERAL_OTHER_TYPES)
820
821/** Mask constant for a UCharCategory. @stable ICU 2.1 */
822#define U_GC_LU_MASK    U_MASK(U_UPPERCASE_LETTER)
823/** Mask constant for a UCharCategory. @stable ICU 2.1 */
824#define U_GC_LL_MASK    U_MASK(U_LOWERCASE_LETTER)
825/** Mask constant for a UCharCategory. @stable ICU 2.1 */
826#define U_GC_LT_MASK    U_MASK(U_TITLECASE_LETTER)
827/** Mask constant for a UCharCategory. @stable ICU 2.1 */
828#define U_GC_LM_MASK    U_MASK(U_MODIFIER_LETTER)
829/** Mask constant for a UCharCategory. @stable ICU 2.1 */
830#define U_GC_LO_MASK    U_MASK(U_OTHER_LETTER)
831
832/** Mask constant for a UCharCategory. @stable ICU 2.1 */
833#define U_GC_MN_MASK    U_MASK(U_NON_SPACING_MARK)
834/** Mask constant for a UCharCategory. @stable ICU 2.1 */
835#define U_GC_ME_MASK    U_MASK(U_ENCLOSING_MARK)
836/** Mask constant for a UCharCategory. @stable ICU 2.1 */
837#define U_GC_MC_MASK    U_MASK(U_COMBINING_SPACING_MARK)
838
839/** Mask constant for a UCharCategory. @stable ICU 2.1 */
840#define U_GC_ND_MASK    U_MASK(U_DECIMAL_DIGIT_NUMBER)
841/** Mask constant for a UCharCategory. @stable ICU 2.1 */
842#define U_GC_NL_MASK    U_MASK(U_LETTER_NUMBER)
843/** Mask constant for a UCharCategory. @stable ICU 2.1 */
844#define U_GC_NO_MASK    U_MASK(U_OTHER_NUMBER)
845
846/** Mask constant for a UCharCategory. @stable ICU 2.1 */
847#define U_GC_ZS_MASK    U_MASK(U_SPACE_SEPARATOR)
848/** Mask constant for a UCharCategory. @stable ICU 2.1 */
849#define U_GC_ZL_MASK    U_MASK(U_LINE_SEPARATOR)
850/** Mask constant for a UCharCategory. @stable ICU 2.1 */
851#define U_GC_ZP_MASK    U_MASK(U_PARAGRAPH_SEPARATOR)
852
853/** Mask constant for a UCharCategory. @stable ICU 2.1 */
854#define U_GC_CC_MASK    U_MASK(U_CONTROL_CHAR)
855/** Mask constant for a UCharCategory. @stable ICU 2.1 */
856#define U_GC_CF_MASK    U_MASK(U_FORMAT_CHAR)
857/** Mask constant for a UCharCategory. @stable ICU 2.1 */
858#define U_GC_CO_MASK    U_MASK(U_PRIVATE_USE_CHAR)
859/** Mask constant for a UCharCategory. @stable ICU 2.1 */
860#define U_GC_CS_MASK    U_MASK(U_SURROGATE)
861
862/** Mask constant for a UCharCategory. @stable ICU 2.1 */
863#define U_GC_PD_MASK    U_MASK(U_DASH_PUNCTUATION)
864/** Mask constant for a UCharCategory. @stable ICU 2.1 */
865#define U_GC_PS_MASK    U_MASK(U_START_PUNCTUATION)
866/** Mask constant for a UCharCategory. @stable ICU 2.1 */
867#define U_GC_PE_MASK    U_MASK(U_END_PUNCTUATION)
868/** Mask constant for a UCharCategory. @stable ICU 2.1 */
869#define U_GC_PC_MASK    U_MASK(U_CONNECTOR_PUNCTUATION)
870/** Mask constant for a UCharCategory. @stable ICU 2.1 */
871#define U_GC_PO_MASK    U_MASK(U_OTHER_PUNCTUATION)
872
873/** Mask constant for a UCharCategory. @stable ICU 2.1 */
874#define U_GC_SM_MASK    U_MASK(U_MATH_SYMBOL)
875/** Mask constant for a UCharCategory. @stable ICU 2.1 */
876#define U_GC_SC_MASK    U_MASK(U_CURRENCY_SYMBOL)
877/** Mask constant for a UCharCategory. @stable ICU 2.1 */
878#define U_GC_SK_MASK    U_MASK(U_MODIFIER_SYMBOL)
879/** Mask constant for a UCharCategory. @stable ICU 2.1 */
880#define U_GC_SO_MASK    U_MASK(U_OTHER_SYMBOL)
881
882/** Mask constant for a UCharCategory. @stable ICU 2.1 */
883#define U_GC_PI_MASK    U_MASK(U_INITIAL_PUNCTUATION)
884/** Mask constant for a UCharCategory. @stable ICU 2.1 */
885#define U_GC_PF_MASK    U_MASK(U_FINAL_PUNCTUATION)
886
887
888/** Mask constant for multiple UCharCategory bits (L Letters). @stable ICU 2.1 */
889#define U_GC_L_MASK \
890            (U_GC_LU_MASK|U_GC_LL_MASK|U_GC_LT_MASK|U_GC_LM_MASK|U_GC_LO_MASK)
891
892/** Mask constant for multiple UCharCategory bits (LC Cased Letters). @stable ICU 2.1 */
893#define U_GC_LC_MASK \
894            (U_GC_LU_MASK|U_GC_LL_MASK|U_GC_LT_MASK)
895
896/** Mask constant for multiple UCharCategory bits (M Marks). @stable ICU 2.1 */
897#define U_GC_M_MASK (U_GC_MN_MASK|U_GC_ME_MASK|U_GC_MC_MASK)
898
899/** Mask constant for multiple UCharCategory bits (N Numbers). @stable ICU 2.1 */
900#define U_GC_N_MASK (U_GC_ND_MASK|U_GC_NL_MASK|U_GC_NO_MASK)
901
902/** Mask constant for multiple UCharCategory bits (Z Separators). @stable ICU 2.1 */
903#define U_GC_Z_MASK (U_GC_ZS_MASK|U_GC_ZL_MASK|U_GC_ZP_MASK)
904
905/** Mask constant for multiple UCharCategory bits (C Others). @stable ICU 2.1 */
906#define U_GC_C_MASK \
907            (U_GC_CN_MASK|U_GC_CC_MASK|U_GC_CF_MASK|U_GC_CO_MASK|U_GC_CS_MASK)
908
909/** Mask constant for multiple UCharCategory bits (P Punctuation). @stable ICU 2.1 */
910#define U_GC_P_MASK \
911            (U_GC_PD_MASK|U_GC_PS_MASK|U_GC_PE_MASK|U_GC_PC_MASK|U_GC_PO_MASK| \
912             U_GC_PI_MASK|U_GC_PF_MASK)
913
914/** Mask constant for multiple UCharCategory bits (S Symbols). @stable ICU 2.1 */
915#define U_GC_S_MASK (U_GC_SM_MASK|U_GC_SC_MASK|U_GC_SK_MASK|U_GC_SO_MASK)
916
917/**
918 * This specifies the language directional property of a character set.
919 * @stable ICU 2.0
920 */
921typedef enum UCharDirection {
922    /*
923     * Note: UCharDirection constants and their API comments are parsed by preparseucd.py.
924     * It matches pairs of lines like
925     *     / ** <Unicode 1..3-letter Bidi_Class value> comment... * /
926     *     U_<[A-Z_]+> = <integer>,
927     */
928
929    /** L @stable ICU 2.0 */
930    U_LEFT_TO_RIGHT               = 0,
931    /** R @stable ICU 2.0 */
932    U_RIGHT_TO_LEFT               = 1,
933    /** EN @stable ICU 2.0 */
934    U_EUROPEAN_NUMBER             = 2,
935    /** ES @stable ICU 2.0 */
936    U_EUROPEAN_NUMBER_SEPARATOR   = 3,
937    /** ET @stable ICU 2.0 */
938    U_EUROPEAN_NUMBER_TERMINATOR  = 4,
939    /** AN @stable ICU 2.0 */
940    U_ARABIC_NUMBER               = 5,
941    /** CS @stable ICU 2.0 */
942    U_COMMON_NUMBER_SEPARATOR     = 6,
943    /** B @stable ICU 2.0 */
944    U_BLOCK_SEPARATOR             = 7,
945    /** S @stable ICU 2.0 */
946    U_SEGMENT_SEPARATOR           = 8,
947    /** WS @stable ICU 2.0 */
948    U_WHITE_SPACE_NEUTRAL         = 9,
949    /** ON @stable ICU 2.0 */
950    U_OTHER_NEUTRAL               = 10,
951    /** LRE @stable ICU 2.0 */
952    U_LEFT_TO_RIGHT_EMBEDDING     = 11,
953    /** LRO @stable ICU 2.0 */
954    U_LEFT_TO_RIGHT_OVERRIDE      = 12,
955    /** AL @stable ICU 2.0 */
956    U_RIGHT_TO_LEFT_ARABIC        = 13,
957    /** RLE @stable ICU 2.0 */
958    U_RIGHT_TO_LEFT_EMBEDDING     = 14,
959    /** RLO @stable ICU 2.0 */
960    U_RIGHT_TO_LEFT_OVERRIDE      = 15,
961    /** PDF @stable ICU 2.0 */
962    U_POP_DIRECTIONAL_FORMAT      = 16,
963    /** NSM @stable ICU 2.0 */
964    U_DIR_NON_SPACING_MARK        = 17,
965    /** BN @stable ICU 2.0 */
966    U_BOUNDARY_NEUTRAL            = 18,
967    /** FSI @stable ICU 52 */
968    U_FIRST_STRONG_ISOLATE        = 19,
969    /** LRI @stable ICU 52 */
970    U_LEFT_TO_RIGHT_ISOLATE       = 20,
971    /** RLI @stable ICU 52 */
972    U_RIGHT_TO_LEFT_ISOLATE       = 21,
973    /** PDI @stable ICU 52 */
974    U_POP_DIRECTIONAL_ISOLATE     = 22,
975#ifndef U_HIDE_DEPRECATED_API
976    /**
977     * One more than the highest UCharDirection value.
978     * The highest value is available via u_getIntPropertyMaxValue(UCHAR_BIDI_CLASS).
979     *
980     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
981     */
982    U_CHAR_DIRECTION_COUNT
983#endif  // U_HIDE_DEPRECATED_API
984} UCharDirection;
985
986/**
987 * Bidi Paired Bracket Type constants.
988 *
989 * @see UCHAR_BIDI_PAIRED_BRACKET_TYPE
990 * @stable ICU 52
991 */
992typedef enum UBidiPairedBracketType {
993    /*
994     * Note: UBidiPairedBracketType constants are parsed by preparseucd.py.
995     * It matches lines like
996     *     U_BPT_<Unicode Bidi_Paired_Bracket_Type value name>
997     */
998
999    /** Not a paired bracket. @stable ICU 52 */
1000    U_BPT_NONE,
1001    /** Open paired bracket. @stable ICU 52 */
1002    U_BPT_OPEN,
1003    /** Close paired bracket. @stable ICU 52 */
1004    U_BPT_CLOSE,
1005#ifndef U_HIDE_DEPRECATED_API
1006    /**
1007     * One more than the highest normal UBidiPairedBracketType value.
1008     * The highest value is available via u_getIntPropertyMaxValue(UCHAR_BIDI_PAIRED_BRACKET_TYPE).
1009     *
1010     * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
1011     */
1012    U_BPT_COUNT /* 3 */
1013#endif  // U_HIDE_DEPRECATED_API
1014} UBidiPairedBracketType;
1015
1016/**
1017 * Constants for Unicode blocks, see the Unicode Data file Blocks.txt
1018 * @stable ICU 2.0
1019 */
1020enum UBlockCode {
1021    /*
1022     * Note: UBlockCode constants are parsed by preparseucd.py.
1023     * It matches lines like
1024     *     UBLOCK_<Unicode Block value name> = <integer>,
1025     */
1026
1027    /** New No_Block value in Unicode 4. @stable ICU 2.6 */
1028    UBLOCK_NO_BLOCK = 0, /*[none]*/ /* Special range indicating No_Block */
1029
1030    /** @stable ICU 2.0 */
1031    UBLOCK_BASIC_LATIN = 1, /*[0000]*/
1032
1033    /** @stable ICU 2.0 */
1034    UBLOCK_LATIN_1_SUPPLEMENT=2, /*[0080]*/
1035
1036    /** @stable ICU 2.0 */
1037    UBLOCK_LATIN_EXTENDED_A =3, /*[0100]*/
1038
1039    /** @stable ICU 2.0 */
1040    UBLOCK_LATIN_EXTENDED_B =4, /*[0180]*/
1041
1042    /** @stable ICU 2.0 */
1043    UBLOCK_IPA_EXTENSIONS =5, /*[0250]*/
1044
1045    /** @stable ICU 2.0 */
1046    UBLOCK_SPACING_MODIFIER_LETTERS =6, /*[02B0]*/
1047
1048    /** @stable ICU 2.0 */
1049    UBLOCK_COMBINING_DIACRITICAL_MARKS =7, /*[0300]*/
1050
1051    /**
1052     * Unicode 3.2 renames this block to "Greek and Coptic".
1053     * @stable ICU 2.0
1054     */
1055    UBLOCK_GREEK =8, /*[0370]*/
1056
1057    /** @stable ICU 2.0 */
1058    UBLOCK_CYRILLIC =9, /*[0400]*/
1059
1060    /** @stable ICU 2.0 */
1061    UBLOCK_ARMENIAN =10, /*[0530]*/
1062
1063    /** @stable ICU 2.0 */
1064    UBLOCK_HEBREW =11, /*[0590]*/
1065
1066    /** @stable ICU 2.0 */
1067    UBLOCK_ARABIC =12, /*[0600]*/
1068
1069    /** @stable ICU 2.0 */
1070    UBLOCK_SYRIAC =13, /*[0700]*/
1071
1072    /** @stable ICU 2.0 */
1073    UBLOCK_THAANA =14, /*[0780]*/
1074
1075    /** @stable ICU 2.0 */
1076    UBLOCK_DEVANAGARI =15, /*[0900]*/
1077
1078    /** @stable ICU 2.0 */
1079    UBLOCK_BENGALI =16, /*[0980]*/
1080
1081    /** @stable ICU 2.0 */
1082    UBLOCK_GURMUKHI =17, /*[0A00]*/
1083
1084    /** @stable ICU 2.0 */
1085    UBLOCK_GUJARATI =18, /*[0A80]*/
1086
1087    /** @stable ICU 2.0 */
1088    UBLOCK_ORIYA =19, /*[0B00]*/
1089
1090    /** @stable ICU 2.0 */
1091    UBLOCK_TAMIL =20, /*[0B80]*/
1092
1093    /** @stable ICU 2.0 */
1094    UBLOCK_TELUGU =21, /*[0C00]*/
1095
1096    /** @stable ICU 2.0 */
1097    UBLOCK_KANNADA =22, /*[0C80]*/
1098
1099    /** @stable ICU 2.0 */
1100    UBLOCK_MALAYALAM =23, /*[0D00]*/
1101
1102    /** @stable ICU 2.0 */
1103    UBLOCK_SINHALA =24, /*[0D80]*/
1104
1105    /** @stable ICU 2.0 */
1106    UBLOCK_THAI =25, /*[0E00]*/
1107
1108    /** @stable ICU 2.0 */
1109    UBLOCK_LAO =26, /*[0E80]*/
1110
1111    /** @stable ICU 2.0 */
1112    UBLOCK_TIBETAN =27, /*[0F00]*/
1113
1114    /** @stable ICU 2.0 */
1115    UBLOCK_MYANMAR =28, /*[1000]*/
1116
1117    /** @stable ICU 2.0 */
1118    UBLOCK_GEORGIAN =29, /*[10A0]*/
1119
1120    /** @stable ICU 2.0 */
1121    UBLOCK_HANGUL_JAMO =30, /*[1100]*/
1122
1123    /** @stable ICU 2.0 */
1124    UBLOCK_ETHIOPIC =31, /*[1200]*/
1125
1126    /** @stable ICU 2.0 */
1127    UBLOCK_CHEROKEE =32, /*[13A0]*/
1128
1129    /** @stable ICU 2.0 */
1130    UBLOCK_UNIFIED_CANADIAN_ABORIGINAL_SYLLABICS =33, /*[1400]*/
1131
1132    /** @stable ICU 2.0 */
1133    UBLOCK_OGHAM =34, /*[1680]*/
1134
1135    /** @stable ICU 2.0 */
1136    UBLOCK_RUNIC =35, /*[16A0]*/
1137
1138    /** @stable ICU 2.0 */
1139    UBLOCK_KHMER =36, /*[1780]*/
1140
1141    /** @stable ICU 2.0 */
1142    UBLOCK_MONGOLIAN =37, /*[1800]*/
1143
1144    /** @stable ICU 2.0 */
1145    UBLOCK_LATIN_EXTENDED_ADDITIONAL =38, /*[1E00]*/
1146
1147    /** @stable ICU 2.0 */
1148    UBLOCK_GREEK_EXTENDED =39, /*[1F00]*/
1149
1150    /** @stable ICU 2.0 */
1151    UBLOCK_GENERAL_PUNCTUATION =40, /*[2000]*/
1152
1153    /** @stable ICU 2.0 */
1154    UBLOCK_SUPERSCRIPTS_AND_SUBSCRIPTS =41, /*[2070]*/
1155
1156    /** @stable ICU 2.0 */
1157    UBLOCK_CURRENCY_SYMBOLS =42, /*[20A0]*/
1158
1159    /**
1160     * Unicode 3.2 renames this block to "Combining Diacritical Marks for Symbols".
1161     * @stable ICU 2.0
1162     */
1163    UBLOCK_COMBINING_MARKS_FOR_SYMBOLS =43, /*[20D0]*/
1164
1165    /** @stable ICU 2.0 */
1166    UBLOCK_LETTERLIKE_SYMBOLS =44, /*[2100]*/
1167
1168    /** @stable ICU 2.0 */
1169    UBLOCK_NUMBER_FORMS =45, /*[2150]*/
1170
1171    /** @stable ICU 2.0 */
1172    UBLOCK_ARROWS =46, /*[2190]*/
1173
1174    /** @stable ICU 2.0 */
1175    UBLOCK_MATHEMATICAL_OPERATORS =47, /*[2200]*/
1176
1177    /** @stable ICU 2.0 */
1178    UBLOCK_MISCELLANEOUS_TECHNICAL =48, /*[2300]*/
1179
1180    /** @stable ICU 2.0 */
1181    UBLOCK_CONTROL_PICTURES =49, /*[2400]*/
1182
1183    /** @stable ICU 2.0 */
1184    UBLOCK_OPTICAL_CHARACTER_RECOGNITION =50, /*[2440]*/
1185
1186    /** @stable ICU 2.0 */
1187    UBLOCK_ENCLOSED_ALPHANUMERICS =51, /*[2460]*/
1188
1189    /** @stable ICU 2.0 */
1190    UBLOCK_BOX_DRAWING =52, /*[2500]*/
1191
1192    /** @stable ICU 2.0 */
1193    UBLOCK_BLOCK_ELEMENTS =53, /*[2580]*/
1194
1195    /** @stable ICU 2.0 */
1196    UBLOCK_GEOMETRIC_SHAPES =54, /*[25A0]*/
1197
1198    /** @stable ICU 2.0 */
1199    UBLOCK_MISCELLANEOUS_SYMBOLS =55, /*[2600]*/
1200

Showing the first 1,200 of 4057 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai