codekingpro/portable-devtools
116k
1// © 2016 and later: Unicode, Inc. and others.
2// License & terms of use: http://www.unicode.org/copyright.html
3/*
4**********************************************************************
5* Copyright (C) 1997-2016, International Business Machines
6* Corporation and others. All Rights Reserved.
7**********************************************************************
8*
9* File UCHAR.H
10*
11* Modification History:
12*
13* Date Name Description
14* 04/02/97 aliu Creation.
15* 03/29/99 helena Updated for C APIs.
16* 4/15/99 Madhu Updated for C Implementation and Javadoc
17* 5/20/99 Madhu Added the function u_getVersion()
18* 8/19/1999 srl Upgraded scripts to Unicode 3.0
19* 8/27/1999 schererm UCharDirection constants: U_...
20* 11/11/1999 weiv added u_isalnum(), cleaned comments
21* 01/11/2000 helena Renamed u_getVersion to u_getUnicodeVersion().
22******************************************************************************
23*/
24
25#ifndef UCHAR_H
26#define UCHAR_H
27
28#include "unicode/utypes.h"
29#include "unicode/stringoptions.h"
30#include "unicode/ucpmap.h"
31
32#if !defined(USET_DEFINED) && !defined(U_IN_DOXYGEN)
33
34#define USET_DEFINED
35
36/**
37 * USet is the C API type corresponding to C++ class UnicodeSet.
38 * It is forward-declared here to avoid including unicode/uset.h file if related
39 * APIs are not used.
40 *
41 * @see ucnv_getUnicodeSet
42 * @stable ICU 2.4
43 */
44typedef struct USet USet;
45
46#endif
47
48
49U_CDECL_BEGIN
50
51/*==========================================================================*/
52/* Unicode version number */
53/*==========================================================================*/
54/**
55 * Unicode version number, default for the current ICU version.
56 * The actual Unicode Character Database (UCD) data is stored in uprops.dat
57 * and may be generated from UCD files from a different Unicode version.
58 * Call u_getUnicodeVersion to get the actual Unicode version of the data.
59 *
60 * @see u_getUnicodeVersion
61 * @stable ICU 2.0
62 */
63#define U_UNICODE_VERSION "13.0"
64
65/**
66 * \file
67 * \brief C API: Unicode Properties
68 *
69 * This C API provides low-level access to the Unicode Character Database.
70 * In addition to raw property values, some convenience functions calculate
71 * derived properties, for example for Java-style programming.
72 *
73 * Unicode assigns each code point (not just assigned character) values for
74 * many properties.
75 * Most of them are simple boolean flags, or constants from a small enumerated list.
76 * For some properties, values are strings or other relatively more complex types.
77 *
78 * For more information see
79 * "About the Unicode Character Database" (http://www.unicode.org/ucd/)
80 * and the ICU User Guide chapter on Properties (http://icu-project.org/userguide/properties.html).
81 *
82 * Many properties are accessible via generic functions that take a UProperty selector.
83 * - u_hasBinaryProperty() returns a binary value (TRUE/FALSE) per property and code point.
84 * - u_getIntPropertyValue() returns an integer value per property and code point.
85 * For each supported enumerated or catalog property, there is
86 * an enum type for all of the property's values, and
87 * u_getIntPropertyValue() returns the numeric values of those constants.
88 * - u_getBinaryPropertySet() returns a set for each ICU-supported binary property with
89 * all code points for which the property is true.
90 * - u_getIntPropertyMap() returns a map for each
91 * ICU-supported enumerated/catalog/int-valued property which
92 * maps all Unicode code points to their values for that property.
93 *
94 * Many functions are designed to match java.lang.Character functions.
95 * See the individual function documentation,
96 * and see the JDK 1.4 java.lang.Character documentation
97 * at http://java.sun.com/j2se/1.4/docs/api/java/lang/Character.html
98 *
99 * There are also functions that provide easy migration from C/POSIX functions
100 * like isblank(). Their use is generally discouraged because the C/POSIX
101 * standards do not define their semantics beyond the ASCII range, which means
102 * that different implementations exhibit very different behavior.
103 * Instead, Unicode properties should be used directly.
104 *
105 * There are also only a few, broad C/POSIX character classes, and they tend
106 * to be used for conflicting purposes. For example, the "isalpha()" class
107 * is sometimes used to determine word boundaries, while a more sophisticated
108 * approach would at least distinguish initial letters from continuation
109 * characters (the latter including combining marks).
110 * (In ICU, BreakIterator is the most sophisticated API for word boundaries.)
111 * Another example: There is no "istitle()" class for titlecase characters.
112 *
113 * ICU 3.4 and later provides API access for all twelve C/POSIX character classes.
114 * ICU implements them according to the Standard Recommendations in
115 * Annex C: Compatibility Properties of UTS #18 Unicode Regular Expressions
116 * (http://www.unicode.org/reports/tr18/#Compatibility_Properties).
117 *
118 * API access for C/POSIX character classes is as follows:
119 * - alpha: u_isUAlphabetic(c) or u_hasBinaryProperty(c, UCHAR_ALPHABETIC)
120 * - lower: u_isULowercase(c) or u_hasBinaryProperty(c, UCHAR_LOWERCASE)
121 * - upper: u_isUUppercase(c) or u_hasBinaryProperty(c, UCHAR_UPPERCASE)
122 * - punct: u_ispunct(c)
123 * - digit: u_isdigit(c) or u_charType(c)==U_DECIMAL_DIGIT_NUMBER
124 * - xdigit: u_isxdigit(c) or u_hasBinaryProperty(c, UCHAR_POSIX_XDIGIT)
125 * - alnum: u_hasBinaryProperty(c, UCHAR_POSIX_ALNUM)
126 * - space: u_isUWhiteSpace(c) or u_hasBinaryProperty(c, UCHAR_WHITE_SPACE)
127 * - blank: u_isblank(c) or u_hasBinaryProperty(c, UCHAR_POSIX_BLANK)
128 * - cntrl: u_charType(c)==U_CONTROL_CHAR
129 * - graph: u_hasBinaryProperty(c, UCHAR_POSIX_GRAPH)
130 * - print: u_hasBinaryProperty(c, UCHAR_POSIX_PRINT)
131 *
132 * Note: Some of the u_isxyz() functions in uchar.h predate, and do not match,
133 * the Standard Recommendations in UTS #18. Instead, they match Java
134 * functions according to their API documentation.
135 *
136 * \htmlonly
137 * The C/POSIX character classes are also available in UnicodeSet patterns,
138 * using patterns like [:graph:] or \p{graph}.
139 * \endhtmlonly
140 *
141 * Note: There are several ICU whitespace functions.
142 * Comparison:
143 * - u_isUWhiteSpace=UCHAR_WHITE_SPACE: Unicode White_Space property;
144 * most of general categories "Z" (separators) + most whitespace ISO controls
145 * (including no-break spaces, but excluding IS1..IS4)
146 * - u_isWhitespace: Java isWhitespace; Z + whitespace ISO controls but excluding no-break spaces
147 * - u_isJavaSpaceChar: Java isSpaceChar; just Z (including no-break spaces)
148 * - u_isspace: Z + whitespace ISO controls (including no-break spaces)
149 * - u_isblank: "horizontal spaces" = TAB + Zs
150 */
151
152/**
153 * Constants.
154 */
155
156/** The lowest Unicode code point value. Code points are non-negative. @stable ICU 2.0 */
157#define UCHAR_MIN_VALUE 0
158
159/**
160 * The highest Unicode code point value (scalar value) according to
161 * The Unicode Standard. This is a 21-bit value (20.1 bits, rounded up).
162 * For a single character, UChar32 is a simple type that can hold any code point value.
163 *
164 * @see UChar32
165 * @stable ICU 2.0
166 */
167#define UCHAR_MAX_VALUE 0x10ffff
168
169/**
170 * Get a single-bit bit set (a flag) from a bit number 0..31.
171 * @stable ICU 2.1
172 */
173#define U_MASK(x) ((uint32_t)1<<(x))
174
175/**
176 * Selection constants for Unicode properties.
177 * These constants are used in functions like u_hasBinaryProperty to select
178 * one of the Unicode properties.
179 *
180 * The properties APIs are intended to reflect Unicode properties as defined
181 * in the Unicode Character Database (UCD) and Unicode Technical Reports (UTR).
182 *
183 * For details about the properties see
184 * UAX #44: Unicode Character Database (http://www.unicode.org/reports/tr44/).
185 *
186 * Important: If ICU is built with UCD files from Unicode versions below, e.g., 3.2,
187 * then properties marked with "new in Unicode 3.2" are not or not fully available.
188 * Check u_getUnicodeVersion to be sure.
189 *
190 * @see u_hasBinaryProperty
191 * @see u_getIntPropertyValue
192 * @see u_getUnicodeVersion
193 * @stable ICU 2.1
194 */
195typedef enum UProperty {
196 /*
197 * Note: UProperty constants are parsed by preparseucd.py.
198 * It matches lines like
199 * UCHAR_<Unicode property name>=<integer>,
200 */
201
202 /* Note: Place UCHAR_ALPHABETIC before UCHAR_BINARY_START so that
203 debuggers display UCHAR_ALPHABETIC as the symbolic name for 0,
204 rather than UCHAR_BINARY_START. Likewise for other *_START
205 identifiers. */
206
207 /** Binary property Alphabetic. Same as u_isUAlphabetic, different from u_isalpha.
208 Lu+Ll+Lt+Lm+Lo+Nl+Other_Alphabetic @stable ICU 2.1 */
209 UCHAR_ALPHABETIC=0,
210 /** First constant for binary Unicode properties. @stable ICU 2.1 */
211 UCHAR_BINARY_START=UCHAR_ALPHABETIC,
212 /** Binary property ASCII_Hex_Digit. 0-9 A-F a-f @stable ICU 2.1 */
213 UCHAR_ASCII_HEX_DIGIT=1,
214 /** Binary property Bidi_Control.
215 Format controls which have specific functions
216 in the Bidi Algorithm. @stable ICU 2.1 */
217 UCHAR_BIDI_CONTROL=2,
218 /** Binary property Bidi_Mirrored.
219 Characters that may change display in RTL text.
220 Same as u_isMirrored.
221 See Bidi Algorithm, UTR 9. @stable ICU 2.1 */
222 UCHAR_BIDI_MIRRORED=3,
223 /** Binary property Dash. Variations of dashes. @stable ICU 2.1 */
224 UCHAR_DASH=4,
225 /** Binary property Default_Ignorable_Code_Point (new in Unicode 3.2).
226 Ignorable in most processing.
227 <2060..206F, FFF0..FFFB, E0000..E0FFF>+Other_Default_Ignorable_Code_Point+(Cf+Cc+Cs-White_Space) @stable ICU 2.1 */
228 UCHAR_DEFAULT_IGNORABLE_CODE_POINT=5,
229 /** Binary property Deprecated (new in Unicode 3.2).
230 The usage of deprecated characters is strongly discouraged. @stable ICU 2.1 */
231 UCHAR_DEPRECATED=6,
232 /** Binary property Diacritic. Characters that linguistically modify
233 the meaning of another character to which they apply. @stable ICU 2.1 */
234 UCHAR_DIACRITIC=7,
235 /** Binary property Extender.
236 Extend the value or shape of a preceding alphabetic character,
237 e.g., length and iteration marks. @stable ICU 2.1 */
238 UCHAR_EXTENDER=8,
239 /** Binary property Full_Composition_Exclusion.
240 CompositionExclusions.txt+Singleton Decompositions+
241 Non-Starter Decompositions. @stable ICU 2.1 */
242 UCHAR_FULL_COMPOSITION_EXCLUSION=9,
243 /** Binary property Grapheme_Base (new in Unicode 3.2).
244 For programmatic determination of grapheme cluster boundaries.
245 [0..10FFFF]-Cc-Cf-Cs-Co-Cn-Zl-Zp-Grapheme_Link-Grapheme_Extend-CGJ @stable ICU 2.1 */
246 UCHAR_GRAPHEME_BASE=10,
247 /** Binary property Grapheme_Extend (new in Unicode 3.2).
248 For programmatic determination of grapheme cluster boundaries.
249 Me+Mn+Mc+Other_Grapheme_Extend-Grapheme_Link-CGJ @stable ICU 2.1 */
250 UCHAR_GRAPHEME_EXTEND=11,
251 /** Binary property Grapheme_Link (new in Unicode 3.2).
252 For programmatic determination of grapheme cluster boundaries. @stable ICU 2.1 */
253 UCHAR_GRAPHEME_LINK=12,
254 /** Binary property Hex_Digit.
255 Characters commonly used for hexadecimal numbers. @stable ICU 2.1 */
256 UCHAR_HEX_DIGIT=13,
257 /** Binary property Hyphen. Dashes used to mark connections
258 between pieces of words, plus the Katakana middle dot. @stable ICU 2.1 */
259 UCHAR_HYPHEN=14,
260 /** Binary property ID_Continue.
261 Characters that can continue an identifier.
262 DerivedCoreProperties.txt also says "NOTE: Cf characters should be filtered out."
263 ID_Start+Mn+Mc+Nd+Pc @stable ICU 2.1 */
264 UCHAR_ID_CONTINUE=15,
265 /** Binary property ID_Start.
266 Characters that can start an identifier.
267 Lu+Ll+Lt+Lm+Lo+Nl @stable ICU 2.1 */
268 UCHAR_ID_START=16,
269 /** Binary property Ideographic.
270 CJKV ideographs. @stable ICU 2.1 */
271 UCHAR_IDEOGRAPHIC=17,
272 /** Binary property IDS_Binary_Operator (new in Unicode 3.2).
273 For programmatic determination of
274 Ideographic Description Sequences. @stable ICU 2.1 */
275 UCHAR_IDS_BINARY_OPERATOR=18,
276 /** Binary property IDS_Trinary_Operator (new in Unicode 3.2).
277 For programmatic determination of
278 Ideographic Description Sequences. @stable ICU 2.1 */
279 UCHAR_IDS_TRINARY_OPERATOR=19,
280 /** Binary property Join_Control.
281 Format controls for cursive joining and ligation. @stable ICU 2.1 */
282 UCHAR_JOIN_CONTROL=20,
283 /** Binary property Logical_Order_Exception (new in Unicode 3.2).
284 Characters that do not use logical order and
285 require special handling in most processing. @stable ICU 2.1 */
286 UCHAR_LOGICAL_ORDER_EXCEPTION=21,
287 /** Binary property Lowercase. Same as u_isULowercase, different from u_islower.
288 Ll+Other_Lowercase @stable ICU 2.1 */
289 UCHAR_LOWERCASE=22,
290 /** Binary property Math. Sm+Other_Math @stable ICU 2.1 */
291 UCHAR_MATH=23,
292 /** Binary property Noncharacter_Code_Point.
293 Code points that are explicitly defined as illegal
294 for the encoding of characters. @stable ICU 2.1 */
295 UCHAR_NONCHARACTER_CODE_POINT=24,
296 /** Binary property Quotation_Mark. @stable ICU 2.1 */
297 UCHAR_QUOTATION_MARK=25,
298 /** Binary property Radical (new in Unicode 3.2).
299 For programmatic determination of
300 Ideographic Description Sequences. @stable ICU 2.1 */
301 UCHAR_RADICAL=26,
302 /** Binary property Soft_Dotted (new in Unicode 3.2).
303 Characters with a "soft dot", like i or j.
304 An accent placed on these characters causes
305 the dot to disappear. @stable ICU 2.1 */
306 UCHAR_SOFT_DOTTED=27,
307 /** Binary property Terminal_Punctuation.
308 Punctuation characters that generally mark
309 the end of textual units. @stable ICU 2.1 */
310 UCHAR_TERMINAL_PUNCTUATION=28,
311 /** Binary property Unified_Ideograph (new in Unicode 3.2).
312 For programmatic determination of
313 Ideographic Description Sequences. @stable ICU 2.1 */
314 UCHAR_UNIFIED_IDEOGRAPH=29,
315 /** Binary property Uppercase. Same as u_isUUppercase, different from u_isupper.
316 Lu+Other_Uppercase @stable ICU 2.1 */
317 UCHAR_UPPERCASE=30,
318 /** Binary property White_Space.
319 Same as u_isUWhiteSpace, different from u_isspace and u_isWhitespace.
320 Space characters+TAB+CR+LF-ZWSP-ZWNBSP @stable ICU 2.1 */
321 UCHAR_WHITE_SPACE=31,
322 /** Binary property XID_Continue.
323 ID_Continue modified to allow closure under
324 normalization forms NFKC and NFKD. @stable ICU 2.1 */
325 UCHAR_XID_CONTINUE=32,
326 /** Binary property XID_Start. ID_Start modified to allow
327 closure under normalization forms NFKC and NFKD. @stable ICU 2.1 */
328 UCHAR_XID_START=33,
329 /** Binary property Case_Sensitive. Either the source of a case
330 mapping or _in_ the target of a case mapping. Not the same as
331 the general category Cased_Letter. @stable ICU 2.6 */
332 UCHAR_CASE_SENSITIVE=34,
333 /** Binary property STerm (new in Unicode 4.0.1).
334 Sentence Terminal. Used in UAX #29: Text Boundaries
335 (http://www.unicode.org/reports/tr29/)
336 @stable ICU 3.0 */
337 UCHAR_S_TERM=35,
338 /** Binary property Variation_Selector (new in Unicode 4.0.1).
339 Indicates all those characters that qualify as Variation Selectors.
340 For details on the behavior of these characters,
341 see StandardizedVariants.html and 15.6 Variation Selectors.
342 @stable ICU 3.0 */
343 UCHAR_VARIATION_SELECTOR=36,
344 /** Binary property NFD_Inert.
345 ICU-specific property for characters that are inert under NFD,
346 i.e., they do not interact with adjacent characters.
347 See the documentation for the Normalizer2 class and the
348 Normalizer2::isInert() method.
349 @stable ICU 3.0 */
350 UCHAR_NFD_INERT=37,
351 /** Binary property NFKD_Inert.
352 ICU-specific property for characters that are inert under NFKD,
353 i.e., they do not interact with adjacent characters.
354 See the documentation for the Normalizer2 class and the
355 Normalizer2::isInert() method.
356 @stable ICU 3.0 */
357 UCHAR_NFKD_INERT=38,
358 /** Binary property NFC_Inert.
359 ICU-specific property for characters that are inert under NFC,
360 i.e., they do not interact with adjacent characters.
361 See the documentation for the Normalizer2 class and the
362 Normalizer2::isInert() method.
363 @stable ICU 3.0 */
364 UCHAR_NFC_INERT=39,
365 /** Binary property NFKC_Inert.
366 ICU-specific property for characters that are inert under NFKC,
367 i.e., they do not interact with adjacent characters.
368 See the documentation for the Normalizer2 class and the
369 Normalizer2::isInert() method.
370 @stable ICU 3.0 */
371 UCHAR_NFKC_INERT=40,
372 /** Binary Property Segment_Starter.
373 ICU-specific property for characters that are starters in terms of
374 Unicode normalization and combining character sequences.
375 They have ccc=0 and do not occur in non-initial position of the
376 canonical decomposition of any character
377 (like a-umlaut in NFD and a Jamo T in an NFD(Hangul LVT)).
378 ICU uses this property for segmenting a string for generating a set of
379 canonically equivalent strings, e.g. for canonical closure while
380 processing collation tailoring rules.
381 @stable ICU 3.0 */
382 UCHAR_SEGMENT_STARTER=41,
383 /** Binary property Pattern_Syntax (new in Unicode 4.1).
384 See UAX #31 Identifier and Pattern Syntax
385 (http://www.unicode.org/reports/tr31/)
386 @stable ICU 3.4 */
387 UCHAR_PATTERN_SYNTAX=42,
388 /** Binary property Pattern_White_Space (new in Unicode 4.1).
389 See UAX #31 Identifier and Pattern Syntax
390 (http://www.unicode.org/reports/tr31/)
391 @stable ICU 3.4 */
392 UCHAR_PATTERN_WHITE_SPACE=43,
393 /** Binary property alnum (a C/POSIX character class).
394 Implemented according to the UTS #18 Annex C Standard Recommendation.
395 See the uchar.h file documentation.
396 @stable ICU 3.4 */
397 UCHAR_POSIX_ALNUM=44,
398 /** Binary property blank (a C/POSIX character class).
399 Implemented according to the UTS #18 Annex C Standard Recommendation.
400 See the uchar.h file documentation.
401 @stable ICU 3.4 */
402 UCHAR_POSIX_BLANK=45,
403 /** Binary property graph (a C/POSIX character class).
404 Implemented according to the UTS #18 Annex C Standard Recommendation.
405 See the uchar.h file documentation.
406 @stable ICU 3.4 */
407 UCHAR_POSIX_GRAPH=46,
408 /** Binary property print (a C/POSIX character class).
409 Implemented according to the UTS #18 Annex C Standard Recommendation.
410 See the uchar.h file documentation.
411 @stable ICU 3.4 */
412 UCHAR_POSIX_PRINT=47,
413 /** Binary property xdigit (a C/POSIX character class).
414 Implemented according to the UTS #18 Annex C Standard Recommendation.
415 See the uchar.h file documentation.
416 @stable ICU 3.4 */
417 UCHAR_POSIX_XDIGIT=48,
418 /** Binary property Cased. For Lowercase, Uppercase and Titlecase characters. @stable ICU 4.4 */
419 UCHAR_CASED=49,
420 /** Binary property Case_Ignorable. Used in context-sensitive case mappings. @stable ICU 4.4 */
421 UCHAR_CASE_IGNORABLE=50,
422 /** Binary property Changes_When_Lowercased. @stable ICU 4.4 */
423 UCHAR_CHANGES_WHEN_LOWERCASED=51,
424 /** Binary property Changes_When_Uppercased. @stable ICU 4.4 */
425 UCHAR_CHANGES_WHEN_UPPERCASED=52,
426 /** Binary property Changes_When_Titlecased. @stable ICU 4.4 */
427 UCHAR_CHANGES_WHEN_TITLECASED=53,
428 /** Binary property Changes_When_Casefolded. @stable ICU 4.4 */
429 UCHAR_CHANGES_WHEN_CASEFOLDED=54,
430 /** Binary property Changes_When_Casemapped. @stable ICU 4.4 */
431 UCHAR_CHANGES_WHEN_CASEMAPPED=55,
432 /** Binary property Changes_When_NFKC_Casefolded. @stable ICU 4.4 */
433 UCHAR_CHANGES_WHEN_NFKC_CASEFOLDED=56,
434 /**
435 * Binary property Emoji.
436 * See http://www.unicode.org/reports/tr51/#Emoji_Properties
437 *
438 * @stable ICU 57
439 */
440 UCHAR_EMOJI=57,
441 /**
442 * Binary property Emoji_Presentation.
443 * See http://www.unicode.org/reports/tr51/#Emoji_Properties
444 *
445 * @stable ICU 57
446 */
447 UCHAR_EMOJI_PRESENTATION=58,
448 /**
449 * Binary property Emoji_Modifier.
450 * See http://www.unicode.org/reports/tr51/#Emoji_Properties
451 *
452 * @stable ICU 57
453 */
454 UCHAR_EMOJI_MODIFIER=59,
455 /**
456 * Binary property Emoji_Modifier_Base.
457 * See http://www.unicode.org/reports/tr51/#Emoji_Properties
458 *
459 * @stable ICU 57
460 */
461 UCHAR_EMOJI_MODIFIER_BASE=60,
462 /**
463 * Binary property Emoji_Component.
464 * See http://www.unicode.org/reports/tr51/#Emoji_Properties
465 *
466 * @stable ICU 60
467 */
468 UCHAR_EMOJI_COMPONENT=61,
469 /**
470 * Binary property Regional_Indicator.
471 * @stable ICU 60
472 */
473 UCHAR_REGIONAL_INDICATOR=62,
474 /**
475 * Binary property Prepended_Concatenation_Mark.
476 * @stable ICU 60
477 */
478 UCHAR_PREPENDED_CONCATENATION_MARK=63,
479 /**
480 * Binary property Extended_Pictographic.
481 * See http://www.unicode.org/reports/tr51/#Emoji_Properties
482 *
483 * @stable ICU 62
484 */
485 UCHAR_EXTENDED_PICTOGRAPHIC=64,
486#ifndef U_HIDE_DEPRECATED_API
487 /**
488 * One more than the last constant for binary Unicode properties.
489 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
490 */
491 UCHAR_BINARY_LIMIT,
492#endif // U_HIDE_DEPRECATED_API
493
494 /** Enumerated property Bidi_Class.
495 Same as u_charDirection, returns UCharDirection values. @stable ICU 2.2 */
496 UCHAR_BIDI_CLASS=0x1000,
497 /** First constant for enumerated/integer Unicode properties. @stable ICU 2.2 */
498 UCHAR_INT_START=UCHAR_BIDI_CLASS,
499 /** Enumerated property Block.
500 Same as ublock_getCode, returns UBlockCode values. @stable ICU 2.2 */
501 UCHAR_BLOCK=0x1001,
502 /** Enumerated property Canonical_Combining_Class.
503 Same as u_getCombiningClass, returns 8-bit numeric values. @stable ICU 2.2 */
504 UCHAR_CANONICAL_COMBINING_CLASS=0x1002,
505 /** Enumerated property Decomposition_Type.
506 Returns UDecompositionType values. @stable ICU 2.2 */
507 UCHAR_DECOMPOSITION_TYPE=0x1003,
508 /** Enumerated property East_Asian_Width.
509 See http://www.unicode.org/reports/tr11/
510 Returns UEastAsianWidth values. @stable ICU 2.2 */
511 UCHAR_EAST_ASIAN_WIDTH=0x1004,
512 /** Enumerated property General_Category.
513 Same as u_charType, returns UCharCategory values. @stable ICU 2.2 */
514 UCHAR_GENERAL_CATEGORY=0x1005,
515 /** Enumerated property Joining_Group.
516 Returns UJoiningGroup values. @stable ICU 2.2 */
517 UCHAR_JOINING_GROUP=0x1006,
518 /** Enumerated property Joining_Type.
519 Returns UJoiningType values. @stable ICU 2.2 */
520 UCHAR_JOINING_TYPE=0x1007,
521 /** Enumerated property Line_Break.
522 Returns ULineBreak values. @stable ICU 2.2 */
523 UCHAR_LINE_BREAK=0x1008,
524 /** Enumerated property Numeric_Type.
525 Returns UNumericType values. @stable ICU 2.2 */
526 UCHAR_NUMERIC_TYPE=0x1009,
527 /** Enumerated property Script.
528 Same as uscript_getScript, returns UScriptCode values. @stable ICU 2.2 */
529 UCHAR_SCRIPT=0x100A,
530 /** Enumerated property Hangul_Syllable_Type, new in Unicode 4.
531 Returns UHangulSyllableType values. @stable ICU 2.6 */
532 UCHAR_HANGUL_SYLLABLE_TYPE=0x100B,
533 /** Enumerated property NFD_Quick_Check.
534 Returns UNormalizationCheckResult values. @stable ICU 3.0 */
535 UCHAR_NFD_QUICK_CHECK=0x100C,
536 /** Enumerated property NFKD_Quick_Check.
537 Returns UNormalizationCheckResult values. @stable ICU 3.0 */
538 UCHAR_NFKD_QUICK_CHECK=0x100D,
539 /** Enumerated property NFC_Quick_Check.
540 Returns UNormalizationCheckResult values. @stable ICU 3.0 */
541 UCHAR_NFC_QUICK_CHECK=0x100E,
542 /** Enumerated property NFKC_Quick_Check.
543 Returns UNormalizationCheckResult values. @stable ICU 3.0 */
544 UCHAR_NFKC_QUICK_CHECK=0x100F,
545 /** Enumerated property Lead_Canonical_Combining_Class.
546 ICU-specific property for the ccc of the first code point
547 of the decomposition, or lccc(c)=ccc(NFD(c)[0]).
548 Useful for checking for canonically ordered text;
549 see UNORM_FCD and http://www.unicode.org/notes/tn5/#FCD .
550 Returns 8-bit numeric values like UCHAR_CANONICAL_COMBINING_CLASS. @stable ICU 3.0 */
551 UCHAR_LEAD_CANONICAL_COMBINING_CLASS=0x1010,
552 /** Enumerated property Trail_Canonical_Combining_Class.
553 ICU-specific property for the ccc of the last code point
554 of the decomposition, or tccc(c)=ccc(NFD(c)[last]).
555 Useful for checking for canonically ordered text;
556 see UNORM_FCD and http://www.unicode.org/notes/tn5/#FCD .
557 Returns 8-bit numeric values like UCHAR_CANONICAL_COMBINING_CLASS. @stable ICU 3.0 */
558 UCHAR_TRAIL_CANONICAL_COMBINING_CLASS=0x1011,
559 /** Enumerated property Grapheme_Cluster_Break (new in Unicode 4.1).
560 Used in UAX #29: Text Boundaries
561 (http://www.unicode.org/reports/tr29/)
562 Returns UGraphemeClusterBreak values. @stable ICU 3.4 */
563 UCHAR_GRAPHEME_CLUSTER_BREAK=0x1012,
564 /** Enumerated property Sentence_Break (new in Unicode 4.1).
565 Used in UAX #29: Text Boundaries
566 (http://www.unicode.org/reports/tr29/)
567 Returns USentenceBreak values. @stable ICU 3.4 */
568 UCHAR_SENTENCE_BREAK=0x1013,
569 /** Enumerated property Word_Break (new in Unicode 4.1).
570 Used in UAX #29: Text Boundaries
571 (http://www.unicode.org/reports/tr29/)
572 Returns UWordBreakValues values. @stable ICU 3.4 */
573 UCHAR_WORD_BREAK=0x1014,
574 /** Enumerated property Bidi_Paired_Bracket_Type (new in Unicode 6.3).
575 Used in UAX #9: Unicode Bidirectional Algorithm
576 (http://www.unicode.org/reports/tr9/)
577 Returns UBidiPairedBracketType values. @stable ICU 52 */
578 UCHAR_BIDI_PAIRED_BRACKET_TYPE=0x1015,
579 /**
580 * Enumerated property Indic_Positional_Category.
581 * New in Unicode 6.0 as provisional property Indic_Matra_Category;
582 * renamed and changed to informative in Unicode 8.0.
583 * See http://www.unicode.org/reports/tr44/#IndicPositionalCategory.txt
584 * @stable ICU 63
585 */
586 UCHAR_INDIC_POSITIONAL_CATEGORY=0x1016,
587 /**
588 * Enumerated property Indic_Syllabic_Category.
589 * New in Unicode 6.0 as provisional; informative since Unicode 8.0.
590 * See http://www.unicode.org/reports/tr44/#IndicSyllabicCategory.txt
591 * @stable ICU 63
592 */
593 UCHAR_INDIC_SYLLABIC_CATEGORY=0x1017,
594 /**
595 * Enumerated property Vertical_Orientation.
596 * Used for UAX #50 Unicode Vertical Text Layout (https://www.unicode.org/reports/tr50/).
597 * New as a UCD property in Unicode 10.0.
598 * @stable ICU 63
599 */
600 UCHAR_VERTICAL_ORIENTATION=0x1018,
601#ifndef U_HIDE_DEPRECATED_API
602 /**
603 * One more than the last constant for enumerated/integer Unicode properties.
604 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
605 */
606 UCHAR_INT_LIMIT=0x1019,
607#endif // U_HIDE_DEPRECATED_API
608
609 /** Bitmask property General_Category_Mask.
610 This is the General_Category property returned as a bit mask.
611 When used in u_getIntPropertyValue(c), same as U_MASK(u_charType(c)),
612 returns bit masks for UCharCategory values where exactly one bit is set.
613 When used with u_getPropertyValueName() and u_getPropertyValueEnum(),
614 a multi-bit mask is used for sets of categories like "Letters".
615 Mask values should be cast to uint32_t.
616 @stable ICU 2.4 */
617 UCHAR_GENERAL_CATEGORY_MASK=0x2000,
618 /** First constant for bit-mask Unicode properties. @stable ICU 2.4 */
619 UCHAR_MASK_START=UCHAR_GENERAL_CATEGORY_MASK,
620#ifndef U_HIDE_DEPRECATED_API
621 /**
622 * One more than the last constant for bit-mask Unicode properties.
623 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
624 */
625 UCHAR_MASK_LIMIT=0x2001,
626#endif // U_HIDE_DEPRECATED_API
627
628 /** Double property Numeric_Value.
629 Corresponds to u_getNumericValue. @stable ICU 2.4 */
630 UCHAR_NUMERIC_VALUE=0x3000,
631 /** First constant for double Unicode properties. @stable ICU 2.4 */
632 UCHAR_DOUBLE_START=UCHAR_NUMERIC_VALUE,
633#ifndef U_HIDE_DEPRECATED_API
634 /**
635 * One more than the last constant for double Unicode properties.
636 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
637 */
638 UCHAR_DOUBLE_LIMIT=0x3001,
639#endif // U_HIDE_DEPRECATED_API
640
641 /** String property Age.
642 Corresponds to u_charAge. @stable ICU 2.4 */
643 UCHAR_AGE=0x4000,
644 /** First constant for string Unicode properties. @stable ICU 2.4 */
645 UCHAR_STRING_START=UCHAR_AGE,
646 /** String property Bidi_Mirroring_Glyph.
647 Corresponds to u_charMirror. @stable ICU 2.4 */
648 UCHAR_BIDI_MIRRORING_GLYPH=0x4001,
649 /** String property Case_Folding.
650 Corresponds to u_strFoldCase in ustring.h. @stable ICU 2.4 */
651 UCHAR_CASE_FOLDING=0x4002,
652#ifndef U_HIDE_DEPRECATED_API
653 /** Deprecated string property ISO_Comment.
654 Corresponds to u_getISOComment. @deprecated ICU 49 */
655 UCHAR_ISO_COMMENT=0x4003,
656#endif /* U_HIDE_DEPRECATED_API */
657 /** String property Lowercase_Mapping.
658 Corresponds to u_strToLower in ustring.h. @stable ICU 2.4 */
659 UCHAR_LOWERCASE_MAPPING=0x4004,
660 /** String property Name.
661 Corresponds to u_charName. @stable ICU 2.4 */
662 UCHAR_NAME=0x4005,
663 /** String property Simple_Case_Folding.
664 Corresponds to u_foldCase. @stable ICU 2.4 */
665 UCHAR_SIMPLE_CASE_FOLDING=0x4006,
666 /** String property Simple_Lowercase_Mapping.
667 Corresponds to u_tolower. @stable ICU 2.4 */
668 UCHAR_SIMPLE_LOWERCASE_MAPPING=0x4007,
669 /** String property Simple_Titlecase_Mapping.
670 Corresponds to u_totitle. @stable ICU 2.4 */
671 UCHAR_SIMPLE_TITLECASE_MAPPING=0x4008,
672 /** String property Simple_Uppercase_Mapping.
673 Corresponds to u_toupper. @stable ICU 2.4 */
674 UCHAR_SIMPLE_UPPERCASE_MAPPING=0x4009,
675 /** String property Titlecase_Mapping.
676 Corresponds to u_strToTitle in ustring.h. @stable ICU 2.4 */
677 UCHAR_TITLECASE_MAPPING=0x400A,
678#ifndef U_HIDE_DEPRECATED_API
679 /** String property Unicode_1_Name.
680 This property is of little practical value.
681 Beginning with ICU 49, ICU APIs return an empty string for this property.
682 Corresponds to u_charName(U_UNICODE_10_CHAR_NAME). @deprecated ICU 49 */
683 UCHAR_UNICODE_1_NAME=0x400B,
684#endif /* U_HIDE_DEPRECATED_API */
685 /** String property Uppercase_Mapping.
686 Corresponds to u_strToUpper in ustring.h. @stable ICU 2.4 */
687 UCHAR_UPPERCASE_MAPPING=0x400C,
688 /** String property Bidi_Paired_Bracket (new in Unicode 6.3).
689 Corresponds to u_getBidiPairedBracket. @stable ICU 52 */
690 UCHAR_BIDI_PAIRED_BRACKET=0x400D,
691#ifndef U_HIDE_DEPRECATED_API
692 /**
693 * One more than the last constant for string Unicode properties.
694 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
695 */
696 UCHAR_STRING_LIMIT=0x400E,
697#endif // U_HIDE_DEPRECATED_API
698
699 /** Miscellaneous property Script_Extensions (new in Unicode 6.0).
700 Some characters are commonly used in multiple scripts.
701 For more information, see UAX #24: http://www.unicode.org/reports/tr24/.
702 Corresponds to uscript_hasScript and uscript_getScriptExtensions in uscript.h.
703 @stable ICU 4.6 */
704 UCHAR_SCRIPT_EXTENSIONS=0x7000,
705 /** First constant for Unicode properties with unusual value types. @stable ICU 4.6 */
706 UCHAR_OTHER_PROPERTY_START=UCHAR_SCRIPT_EXTENSIONS,
707#ifndef U_HIDE_DEPRECATED_API
708 /**
709 * One more than the last constant for Unicode properties with unusual value types.
710 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
711 */
712 UCHAR_OTHER_PROPERTY_LIMIT=0x7001,
713#endif // U_HIDE_DEPRECATED_API
714
715 /** Represents a nonexistent or invalid property or property value. @stable ICU 2.4 */
716 UCHAR_INVALID_CODE = -1
717} UProperty;
718
719/**
720 * Data for enumerated Unicode general category types.
721 * See http://www.unicode.org/Public/UNIDATA/UnicodeData.html .
722 * @stable ICU 2.0
723 */
724typedef enum UCharCategory
725{
726 /*
727 * Note: UCharCategory constants and their API comments are parsed by preparseucd.py.
728 * It matches pairs of lines like
729 * / ** <Unicode 2-letter General_Category value> comment... * /
730 * U_<[A-Z_]+> = <integer>,
731 */
732
733 /** Non-category for unassigned and non-character code points. @stable ICU 2.0 */
734 U_UNASSIGNED = 0,
735 /** Cn "Other, Not Assigned (no characters in [UnicodeData.txt] have this property)" (same as U_UNASSIGNED!) @stable ICU 2.0 */
736 U_GENERAL_OTHER_TYPES = 0,
737 /** Lu @stable ICU 2.0 */
738 U_UPPERCASE_LETTER = 1,
739 /** Ll @stable ICU 2.0 */
740 U_LOWERCASE_LETTER = 2,
741 /** Lt @stable ICU 2.0 */
742 U_TITLECASE_LETTER = 3,
743 /** Lm @stable ICU 2.0 */
744 U_MODIFIER_LETTER = 4,
745 /** Lo @stable ICU 2.0 */
746 U_OTHER_LETTER = 5,
747 /** Mn @stable ICU 2.0 */
748 U_NON_SPACING_MARK = 6,
749 /** Me @stable ICU 2.0 */
750 U_ENCLOSING_MARK = 7,
751 /** Mc @stable ICU 2.0 */
752 U_COMBINING_SPACING_MARK = 8,
753 /** Nd @stable ICU 2.0 */
754 U_DECIMAL_DIGIT_NUMBER = 9,
755 /** Nl @stable ICU 2.0 */
756 U_LETTER_NUMBER = 10,
757 /** No @stable ICU 2.0 */
758 U_OTHER_NUMBER = 11,
759 /** Zs @stable ICU 2.0 */
760 U_SPACE_SEPARATOR = 12,
761 /** Zl @stable ICU 2.0 */
762 U_LINE_SEPARATOR = 13,
763 /** Zp @stable ICU 2.0 */
764 U_PARAGRAPH_SEPARATOR = 14,
765 /** Cc @stable ICU 2.0 */
766 U_CONTROL_CHAR = 15,
767 /** Cf @stable ICU 2.0 */
768 U_FORMAT_CHAR = 16,
769 /** Co @stable ICU 2.0 */
770 U_PRIVATE_USE_CHAR = 17,
771 /** Cs @stable ICU 2.0 */
772 U_SURROGATE = 18,
773 /** Pd @stable ICU 2.0 */
774 U_DASH_PUNCTUATION = 19,
775 /** Ps @stable ICU 2.0 */
776 U_START_PUNCTUATION = 20,
777 /** Pe @stable ICU 2.0 */
778 U_END_PUNCTUATION = 21,
779 /** Pc @stable ICU 2.0 */
780 U_CONNECTOR_PUNCTUATION = 22,
781 /** Po @stable ICU 2.0 */
782 U_OTHER_PUNCTUATION = 23,
783 /** Sm @stable ICU 2.0 */
784 U_MATH_SYMBOL = 24,
785 /** Sc @stable ICU 2.0 */
786 U_CURRENCY_SYMBOL = 25,
787 /** Sk @stable ICU 2.0 */
788 U_MODIFIER_SYMBOL = 26,
789 /** So @stable ICU 2.0 */
790 U_OTHER_SYMBOL = 27,
791 /** Pi @stable ICU 2.0 */
792 U_INITIAL_PUNCTUATION = 28,
793 /** Pf @stable ICU 2.0 */
794 U_FINAL_PUNCTUATION = 29,
795 /**
796 * One higher than the last enum UCharCategory constant.
797 * This numeric value is stable (will not change), see
798 * http://www.unicode.org/policies/stability_policy.html#Property_Value
799 *
800 * @stable ICU 2.0
801 */
802 U_CHAR_CATEGORY_COUNT
803} UCharCategory;
804
805/**
806 * U_GC_XX_MASK constants are bit flags corresponding to Unicode
807 * general category values.
808 * For each category, the nth bit is set if the numeric value of the
809 * corresponding UCharCategory constant is n.
810 *
811 * There are also some U_GC_Y_MASK constants for groups of general categories
812 * like L for all letter categories.
813 *
814 * @see u_charType
815 * @see U_GET_GC_MASK
816 * @see UCharCategory
817 * @stable ICU 2.1
818 */
819#define U_GC_CN_MASK U_MASK(U_GENERAL_OTHER_TYPES)
820
821/** Mask constant for a UCharCategory. @stable ICU 2.1 */
822#define U_GC_LU_MASK U_MASK(U_UPPERCASE_LETTER)
823/** Mask constant for a UCharCategory. @stable ICU 2.1 */
824#define U_GC_LL_MASK U_MASK(U_LOWERCASE_LETTER)
825/** Mask constant for a UCharCategory. @stable ICU 2.1 */
826#define U_GC_LT_MASK U_MASK(U_TITLECASE_LETTER)
827/** Mask constant for a UCharCategory. @stable ICU 2.1 */
828#define U_GC_LM_MASK U_MASK(U_MODIFIER_LETTER)
829/** Mask constant for a UCharCategory. @stable ICU 2.1 */
830#define U_GC_LO_MASK U_MASK(U_OTHER_LETTER)
831
832/** Mask constant for a UCharCategory. @stable ICU 2.1 */
833#define U_GC_MN_MASK U_MASK(U_NON_SPACING_MARK)
834/** Mask constant for a UCharCategory. @stable ICU 2.1 */
835#define U_GC_ME_MASK U_MASK(U_ENCLOSING_MARK)
836/** Mask constant for a UCharCategory. @stable ICU 2.1 */
837#define U_GC_MC_MASK U_MASK(U_COMBINING_SPACING_MARK)
838
839/** Mask constant for a UCharCategory. @stable ICU 2.1 */
840#define U_GC_ND_MASK U_MASK(U_DECIMAL_DIGIT_NUMBER)
841/** Mask constant for a UCharCategory. @stable ICU 2.1 */
842#define U_GC_NL_MASK U_MASK(U_LETTER_NUMBER)
843/** Mask constant for a UCharCategory. @stable ICU 2.1 */
844#define U_GC_NO_MASK U_MASK(U_OTHER_NUMBER)
845
846/** Mask constant for a UCharCategory. @stable ICU 2.1 */
847#define U_GC_ZS_MASK U_MASK(U_SPACE_SEPARATOR)
848/** Mask constant for a UCharCategory. @stable ICU 2.1 */
849#define U_GC_ZL_MASK U_MASK(U_LINE_SEPARATOR)
850/** Mask constant for a UCharCategory. @stable ICU 2.1 */
851#define U_GC_ZP_MASK U_MASK(U_PARAGRAPH_SEPARATOR)
852
853/** Mask constant for a UCharCategory. @stable ICU 2.1 */
854#define U_GC_CC_MASK U_MASK(U_CONTROL_CHAR)
855/** Mask constant for a UCharCategory. @stable ICU 2.1 */
856#define U_GC_CF_MASK U_MASK(U_FORMAT_CHAR)
857/** Mask constant for a UCharCategory. @stable ICU 2.1 */
858#define U_GC_CO_MASK U_MASK(U_PRIVATE_USE_CHAR)
859/** Mask constant for a UCharCategory. @stable ICU 2.1 */
860#define U_GC_CS_MASK U_MASK(U_SURROGATE)
861
862/** Mask constant for a UCharCategory. @stable ICU 2.1 */
863#define U_GC_PD_MASK U_MASK(U_DASH_PUNCTUATION)
864/** Mask constant for a UCharCategory. @stable ICU 2.1 */
865#define U_GC_PS_MASK U_MASK(U_START_PUNCTUATION)
866/** Mask constant for a UCharCategory. @stable ICU 2.1 */
867#define U_GC_PE_MASK U_MASK(U_END_PUNCTUATION)
868/** Mask constant for a UCharCategory. @stable ICU 2.1 */
869#define U_GC_PC_MASK U_MASK(U_CONNECTOR_PUNCTUATION)
870/** Mask constant for a UCharCategory. @stable ICU 2.1 */
871#define U_GC_PO_MASK U_MASK(U_OTHER_PUNCTUATION)
872
873/** Mask constant for a UCharCategory. @stable ICU 2.1 */
874#define U_GC_SM_MASK U_MASK(U_MATH_SYMBOL)
875/** Mask constant for a UCharCategory. @stable ICU 2.1 */
876#define U_GC_SC_MASK U_MASK(U_CURRENCY_SYMBOL)
877/** Mask constant for a UCharCategory. @stable ICU 2.1 */
878#define U_GC_SK_MASK U_MASK(U_MODIFIER_SYMBOL)
879/** Mask constant for a UCharCategory. @stable ICU 2.1 */
880#define U_GC_SO_MASK U_MASK(U_OTHER_SYMBOL)
881
882/** Mask constant for a UCharCategory. @stable ICU 2.1 */
883#define U_GC_PI_MASK U_MASK(U_INITIAL_PUNCTUATION)
884/** Mask constant for a UCharCategory. @stable ICU 2.1 */
885#define U_GC_PF_MASK U_MASK(U_FINAL_PUNCTUATION)
886
887
888/** Mask constant for multiple UCharCategory bits (L Letters). @stable ICU 2.1 */
889#define U_GC_L_MASK \
890 (U_GC_LU_MASK|U_GC_LL_MASK|U_GC_LT_MASK|U_GC_LM_MASK|U_GC_LO_MASK)
891
892/** Mask constant for multiple UCharCategory bits (LC Cased Letters). @stable ICU 2.1 */
893#define U_GC_LC_MASK \
894 (U_GC_LU_MASK|U_GC_LL_MASK|U_GC_LT_MASK)
895
896/** Mask constant for multiple UCharCategory bits (M Marks). @stable ICU 2.1 */
897#define U_GC_M_MASK (U_GC_MN_MASK|U_GC_ME_MASK|U_GC_MC_MASK)
898
899/** Mask constant for multiple UCharCategory bits (N Numbers). @stable ICU 2.1 */
900#define U_GC_N_MASK (U_GC_ND_MASK|U_GC_NL_MASK|U_GC_NO_MASK)
901
902/** Mask constant for multiple UCharCategory bits (Z Separators). @stable ICU 2.1 */
903#define U_GC_Z_MASK (U_GC_ZS_MASK|U_GC_ZL_MASK|U_GC_ZP_MASK)
904
905/** Mask constant for multiple UCharCategory bits (C Others). @stable ICU 2.1 */
906#define U_GC_C_MASK \
907 (U_GC_CN_MASK|U_GC_CC_MASK|U_GC_CF_MASK|U_GC_CO_MASK|U_GC_CS_MASK)
908
909/** Mask constant for multiple UCharCategory bits (P Punctuation). @stable ICU 2.1 */
910#define U_GC_P_MASK \
911 (U_GC_PD_MASK|U_GC_PS_MASK|U_GC_PE_MASK|U_GC_PC_MASK|U_GC_PO_MASK| \
912 U_GC_PI_MASK|U_GC_PF_MASK)
913
914/** Mask constant for multiple UCharCategory bits (S Symbols). @stable ICU 2.1 */
915#define U_GC_S_MASK (U_GC_SM_MASK|U_GC_SC_MASK|U_GC_SK_MASK|U_GC_SO_MASK)
916
917/**
918 * This specifies the language directional property of a character set.
919 * @stable ICU 2.0
920 */
921typedef enum UCharDirection {
922 /*
923 * Note: UCharDirection constants and their API comments are parsed by preparseucd.py.
924 * It matches pairs of lines like
925 * / ** <Unicode 1..3-letter Bidi_Class value> comment... * /
926 * U_<[A-Z_]+> = <integer>,
927 */
928
929 /** L @stable ICU 2.0 */
930 U_LEFT_TO_RIGHT = 0,
931 /** R @stable ICU 2.0 */
932 U_RIGHT_TO_LEFT = 1,
933 /** EN @stable ICU 2.0 */
934 U_EUROPEAN_NUMBER = 2,
935 /** ES @stable ICU 2.0 */
936 U_EUROPEAN_NUMBER_SEPARATOR = 3,
937 /** ET @stable ICU 2.0 */
938 U_EUROPEAN_NUMBER_TERMINATOR = 4,
939 /** AN @stable ICU 2.0 */
940 U_ARABIC_NUMBER = 5,
941 /** CS @stable ICU 2.0 */
942 U_COMMON_NUMBER_SEPARATOR = 6,
943 /** B @stable ICU 2.0 */
944 U_BLOCK_SEPARATOR = 7,
945 /** S @stable ICU 2.0 */
946 U_SEGMENT_SEPARATOR = 8,
947 /** WS @stable ICU 2.0 */
948 U_WHITE_SPACE_NEUTRAL = 9,
949 /** ON @stable ICU 2.0 */
950 U_OTHER_NEUTRAL = 10,
951 /** LRE @stable ICU 2.0 */
952 U_LEFT_TO_RIGHT_EMBEDDING = 11,
953 /** LRO @stable ICU 2.0 */
954 U_LEFT_TO_RIGHT_OVERRIDE = 12,
955 /** AL @stable ICU 2.0 */
956 U_RIGHT_TO_LEFT_ARABIC = 13,
957 /** RLE @stable ICU 2.0 */
958 U_RIGHT_TO_LEFT_EMBEDDING = 14,
959 /** RLO @stable ICU 2.0 */
960 U_RIGHT_TO_LEFT_OVERRIDE = 15,
961 /** PDF @stable ICU 2.0 */
962 U_POP_DIRECTIONAL_FORMAT = 16,
963 /** NSM @stable ICU 2.0 */
964 U_DIR_NON_SPACING_MARK = 17,
965 /** BN @stable ICU 2.0 */
966 U_BOUNDARY_NEUTRAL = 18,
967 /** FSI @stable ICU 52 */
968 U_FIRST_STRONG_ISOLATE = 19,
969 /** LRI @stable ICU 52 */
970 U_LEFT_TO_RIGHT_ISOLATE = 20,
971 /** RLI @stable ICU 52 */
972 U_RIGHT_TO_LEFT_ISOLATE = 21,
973 /** PDI @stable ICU 52 */
974 U_POP_DIRECTIONAL_ISOLATE = 22,
975#ifndef U_HIDE_DEPRECATED_API
976 /**
977 * One more than the highest UCharDirection value.
978 * The highest value is available via u_getIntPropertyMaxValue(UCHAR_BIDI_CLASS).
979 *
980 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
981 */
982 U_CHAR_DIRECTION_COUNT
983#endif // U_HIDE_DEPRECATED_API
984} UCharDirection;
985
986/**
987 * Bidi Paired Bracket Type constants.
988 *
989 * @see UCHAR_BIDI_PAIRED_BRACKET_TYPE
990 * @stable ICU 52
991 */
992typedef enum UBidiPairedBracketType {
993 /*
994 * Note: UBidiPairedBracketType constants are parsed by preparseucd.py.
995 * It matches lines like
996 * U_BPT_<Unicode Bidi_Paired_Bracket_Type value name>
997 */
998
999 /** Not a paired bracket. @stable ICU 52 */
1000 U_BPT_NONE,
1001 /** Open paired bracket. @stable ICU 52 */
1002 U_BPT_OPEN,
1003 /** Close paired bracket. @stable ICU 52 */
1004 U_BPT_CLOSE,
1005#ifndef U_HIDE_DEPRECATED_API
1006 /**
1007 * One more than the highest normal UBidiPairedBracketType value.
1008 * The highest value is available via u_getIntPropertyMaxValue(UCHAR_BIDI_PAIRED_BRACKET_TYPE).
1009 *
1010 * @deprecated ICU 58 The numeric value may change over time, see ICU ticket #12420.
1011 */
1012 U_BPT_COUNT /* 3 */
1013#endif // U_HIDE_DEPRECATED_API
1014} UBidiPairedBracketType;
1015
1016/**
1017 * Constants for Unicode blocks, see the Unicode Data file Blocks.txt
1018 * @stable ICU 2.0
1019 */
1020enum UBlockCode {
1021 /*
1022 * Note: UBlockCode constants are parsed by preparseucd.py.
1023 * It matches lines like
1024 * UBLOCK_<Unicode Block value name> = <integer>,
1025 */
1026
1027 /** New No_Block value in Unicode 4. @stable ICU 2.6 */
1028 UBLOCK_NO_BLOCK = 0, /*[none]*/ /* Special range indicating No_Block */
1029
1030 /** @stable ICU 2.0 */
1031 UBLOCK_BASIC_LATIN = 1, /*[0000]*/
1032
1033 /** @stable ICU 2.0 */
1034 UBLOCK_LATIN_1_SUPPLEMENT=2, /*[0080]*/
1035
1036 /** @stable ICU 2.0 */
1037 UBLOCK_LATIN_EXTENDED_A =3, /*[0100]*/
1038
1039 /** @stable ICU 2.0 */
1040 UBLOCK_LATIN_EXTENDED_B =4, /*[0180]*/
1041
1042 /** @stable ICU 2.0 */
1043 UBLOCK_IPA_EXTENSIONS =5, /*[0250]*/
1044
1045 /** @stable ICU 2.0 */
1046 UBLOCK_SPACING_MODIFIER_LETTERS =6, /*[02B0]*/
1047
1048 /** @stable ICU 2.0 */
1049 UBLOCK_COMBINING_DIACRITICAL_MARKS =7, /*[0300]*/
1050
1051 /**
1052 * Unicode 3.2 renames this block to "Greek and Coptic".
1053 * @stable ICU 2.0
1054 */
1055 UBLOCK_GREEK =8, /*[0370]*/
1056
1057 /** @stable ICU 2.0 */
1058 UBLOCK_CYRILLIC =9, /*[0400]*/
1059
1060 /** @stable ICU 2.0 */
1061 UBLOCK_ARMENIAN =10, /*[0530]*/
1062
1063 /** @stable ICU 2.0 */
1064 UBLOCK_HEBREW =11, /*[0590]*/
1065
1066 /** @stable ICU 2.0 */
1067 UBLOCK_ARABIC =12, /*[0600]*/
1068
1069 /** @stable ICU 2.0 */
1070 UBLOCK_SYRIAC =13, /*[0700]*/
1071
1072 /** @stable ICU 2.0 */
1073 UBLOCK_THAANA =14, /*[0780]*/
1074
1075 /** @stable ICU 2.0 */
1076 UBLOCK_DEVANAGARI =15, /*[0900]*/
1077
1078 /** @stable ICU 2.0 */
1079 UBLOCK_BENGALI =16, /*[0980]*/
1080
1081 /** @stable ICU 2.0 */
1082 UBLOCK_GURMUKHI =17, /*[0A00]*/
1083
1084 /** @stable ICU 2.0 */
1085 UBLOCK_GUJARATI =18, /*[0A80]*/
1086
1087 /** @stable ICU 2.0 */
1088 UBLOCK_ORIYA =19, /*[0B00]*/
1089
1090 /** @stable ICU 2.0 */
1091 UBLOCK_TAMIL =20, /*[0B80]*/
1092
1093 /** @stable ICU 2.0 */
1094 UBLOCK_TELUGU =21, /*[0C00]*/
1095
1096 /** @stable ICU 2.0 */
1097 UBLOCK_KANNADA =22, /*[0C80]*/
1098
1099 /** @stable ICU 2.0 */
1100 UBLOCK_MALAYALAM =23, /*[0D00]*/
1101
1102 /** @stable ICU 2.0 */
1103 UBLOCK_SINHALA =24, /*[0D80]*/
1104
1105 /** @stable ICU 2.0 */
1106 UBLOCK_THAI =25, /*[0E00]*/
1107
1108 /** @stable ICU 2.0 */
1109 UBLOCK_LAO =26, /*[0E80]*/
1110
1111 /** @stable ICU 2.0 */
1112 UBLOCK_TIBETAN =27, /*[0F00]*/
1113
1114 /** @stable ICU 2.0 */
1115 UBLOCK_MYANMAR =28, /*[1000]*/
1116
1117 /** @stable ICU 2.0 */
1118 UBLOCK_GEORGIAN =29, /*[10A0]*/
1119
1120 /** @stable ICU 2.0 */
1121 UBLOCK_HANGUL_JAMO =30, /*[1100]*/
1122
1123 /** @stable ICU 2.0 */
1124 UBLOCK_ETHIOPIC =31, /*[1200]*/
1125
1126 /** @stable ICU 2.0 */
1127 UBLOCK_CHEROKEE =32, /*[13A0]*/
1128
1129 /** @stable ICU 2.0 */
1130 UBLOCK_UNIFIED_CANADIAN_ABORIGINAL_SYLLABICS =33, /*[1400]*/
1131
1132 /** @stable ICU 2.0 */
1133 UBLOCK_OGHAM =34, /*[1680]*/
1134
1135 /** @stable ICU 2.0 */
1136 UBLOCK_RUNIC =35, /*[16A0]*/
1137
1138 /** @stable ICU 2.0 */
1139 UBLOCK_KHMER =36, /*[1780]*/
1140
1141 /** @stable ICU 2.0 */
1142 UBLOCK_MONGOLIAN =37, /*[1800]*/
1143
1144 /** @stable ICU 2.0 */
1145 UBLOCK_LATIN_EXTENDED_ADDITIONAL =38, /*[1E00]*/
1146
1147 /** @stable ICU 2.0 */
1148 UBLOCK_GREEK_EXTENDED =39, /*[1F00]*/
1149
1150 /** @stable ICU 2.0 */
1151 UBLOCK_GENERAL_PUNCTUATION =40, /*[2000]*/
1152
1153 /** @stable ICU 2.0 */
1154 UBLOCK_SUPERSCRIPTS_AND_SUBSCRIPTS =41, /*[2070]*/
1155
1156 /** @stable ICU 2.0 */
1157 UBLOCK_CURRENCY_SYMBOLS =42, /*[20A0]*/
1158
1159 /**
1160 * Unicode 3.2 renames this block to "Combining Diacritical Marks for Symbols".
1161 * @stable ICU 2.0
1162 */
1163 UBLOCK_COMBINING_MARKS_FOR_SYMBOLS =43, /*[20D0]*/
1164
1165 /** @stable ICU 2.0 */
1166 UBLOCK_LETTERLIKE_SYMBOLS =44, /*[2100]*/
1167
1168 /** @stable ICU 2.0 */
1169 UBLOCK_NUMBER_FORMS =45, /*[2150]*/
1170
1171 /** @stable ICU 2.0 */
1172 UBLOCK_ARROWS =46, /*[2190]*/
1173
1174 /** @stable ICU 2.0 */
1175 UBLOCK_MATHEMATICAL_OPERATORS =47, /*[2200]*/
1176
1177 /** @stable ICU 2.0 */
1178 UBLOCK_MISCELLANEOUS_TECHNICAL =48, /*[2300]*/
1179
1180 /** @stable ICU 2.0 */
1181 UBLOCK_CONTROL_PICTURES =49, /*[2400]*/
1182
1183 /** @stable ICU 2.0 */
1184 UBLOCK_OPTICAL_CHARACTER_RECOGNITION =50, /*[2440]*/
1185
1186 /** @stable ICU 2.0 */
1187 UBLOCK_ENCLOSED_ALPHANUMERICS =51, /*[2460]*/
1188
1189 /** @stable ICU 2.0 */
1190 UBLOCK_BOX_DRAWING =52, /*[2500]*/
1191
1192 /** @stable ICU 2.0 */
1193 UBLOCK_BLOCK_ELEMENTS =53, /*[2580]*/
1194
1195 /** @stable ICU 2.0 */
1196 UBLOCK_GEOMETRIC_SHAPES =54, /*[25A0]*/
1197
1198 /** @stable ICU 2.0 */
1199 UBLOCK_MISCELLANEOUS_SYMBOLS =55, /*[2600]*/
1200
