codekingpro/portable-devtools
115k
1#ifndef Py_CPYTHON_UNICODEOBJECT_H2# error "this header file must not be included directly"3#endif4 5/* Py_UNICODE was the native Unicode storage format (code unit) used by6 Python and represents a single Unicode element in the Unicode type.7 With PEP 393, Py_UNICODE is deprecated and replaced with a8 typedef to wchar_t. */9Py_DEPRECATED(3.13) typedef wchar_t PY_UNICODE_TYPE;10Py_DEPRECATED(3.13) typedef wchar_t Py_UNICODE;11 12 13/* --- Internal Unicode Operations ---------------------------------------- */14 15// Static inline functions to work with surrogates16static inline int Py_UNICODE_IS_SURROGATE(Py_UCS4 ch) {17 return (0xD800 <= ch && ch <= 0xDFFF);18}19static inline int Py_UNICODE_IS_HIGH_SURROGATE(Py_UCS4 ch) {20 return (0xD800 <= ch && ch <= 0xDBFF);21}22static inline int Py_UNICODE_IS_LOW_SURROGATE(Py_UCS4 ch) {23 return (0xDC00 <= ch && ch <= 0xDFFF);24}25 26// Join two surrogate characters and return a single Py_UCS4 value.27static inline Py_UCS4 Py_UNICODE_JOIN_SURROGATES(Py_UCS4 high, Py_UCS4 low) {28 assert(Py_UNICODE_IS_HIGH_SURROGATE(high));29 assert(Py_UNICODE_IS_LOW_SURROGATE(low));30 return 0x10000 + (((high & 0x03FF) << 10) | (low & 0x03FF));31}32 33// High surrogate = top 10 bits added to 0xD800.34// The character must be in the range [U+10000; U+10ffff].35static inline Py_UCS4 Py_UNICODE_HIGH_SURROGATE(Py_UCS4 ch) {36 assert(0x10000 <= ch && ch <= 0x10ffff);37 return (0xD800 - (0x10000 >> 10) + (ch >> 10));38}39 40// Low surrogate = bottom 10 bits added to 0xDC00.41// The character must be in the range [U+10000; U+10ffff].42static inline Py_UCS4 Py_UNICODE_LOW_SURROGATE(Py_UCS4 ch) {43 assert(0x10000 <= ch && ch <= 0x10ffff);44 return (0xDC00 + (ch & 0x3FF));45}46 47 48/* --- Unicode Type ------------------------------------------------------- */49 50/* ASCII-only strings created through PyUnicode_New use the PyASCIIObject51 structure. state.ascii and state.compact are set, and the data52 immediately follow the structure. utf8_length can be found53 in the length field; the utf8 pointer is equal to the data pointer. */54typedef struct {55 /* There are 4 forms of Unicode strings:56 57 - compact ascii:58 59 * structure = PyASCIIObject60 * test: PyUnicode_IS_COMPACT_ASCII(op)61 * kind = PyUnicode_1BYTE_KIND62 * compact = 163 * ascii = 164 * (length is the length of the utf8)65 * (data starts just after the structure)66 * (since ASCII is decoded from UTF-8, the utf8 string are the data)67 68 - compact:69 70 * structure = PyCompactUnicodeObject71 * test: PyUnicode_IS_COMPACT(op) && !PyUnicode_IS_ASCII(op)72 * kind = PyUnicode_1BYTE_KIND, PyUnicode_2BYTE_KIND or73 PyUnicode_4BYTE_KIND74 * compact = 175 * ascii = 076 * utf8 is not shared with data77 * utf8_length = 0 if utf8 is NULL78 * (data starts just after the structure)79 80 - legacy string:81 82 * structure = PyUnicodeObject structure83 * test: !PyUnicode_IS_COMPACT(op)84 * kind = PyUnicode_1BYTE_KIND, PyUnicode_2BYTE_KIND or85 PyUnicode_4BYTE_KIND86 * compact = 087 * data.any is not NULL88 * utf8 is shared and utf8_length = length with data.any if ascii = 189 * utf8_length = 0 if utf8 is NULL90 91 Compact strings use only one memory block (structure + characters),92 whereas legacy strings use one block for the structure and one block93 for characters.94 95 Legacy strings are created by subclasses of Unicode.96 97 See also _PyUnicode_CheckConsistency().98 */99 PyObject_HEAD100 Py_ssize_t length; /* Number of code points in the string */101 Py_hash_t hash; /* Hash value; -1 if not set */102#ifdef Py_GIL_DISABLED103 /* Ensure 4 byte alignment for PyUnicode_DATA(), see gh-63736 on m68k.104 In the non-free-threaded build, we'll use explicit padding instead */105 _Py_ALIGN_AS(4)106#endif107 struct {108 /* If interned is non-zero, the two references from the109 dictionary to this object are *not* counted in ob_refcnt.110 The possible values here are:111 0: Not Interned112 1: Interned113 2: Interned and Immortal114 3: Interned, Immortal, and Static115 This categorization allows the runtime to determine the right116 cleanup mechanism at runtime shutdown. */117#ifdef Py_GIL_DISABLED118 // Needs to be accessed atomically, so can't be a bit field.119 unsigned char interned;120#else121 unsigned int interned:2;122#endif123 /* Character size:124 125 - PyUnicode_1BYTE_KIND (1):126 127 * character type = Py_UCS1 (8 bits, unsigned)128 * all characters are in the range U+0000-U+00FF (latin1)129 * if ascii is set, all characters are in the range U+0000-U+007F130 (ASCII), otherwise at least one character is in the range131 U+0080-U+00FF132 133 - PyUnicode_2BYTE_KIND (2):134 135 * character type = Py_UCS2 (16 bits, unsigned)136 * all characters are in the range U+0000-U+FFFF (BMP)137 * at least one character is in the range U+0100-U+FFFF138 139 - PyUnicode_4BYTE_KIND (4):140 141 * character type = Py_UCS4 (32 bits, unsigned)142 * all characters are in the range U+0000-U+10FFFF143 * at least one character is in the range U+10000-U+10FFFF144 */145 unsigned int kind:3;146 /* Compact is with respect to the allocation scheme. Compact unicode147 objects only require one memory block while non-compact objects use148 one block for the PyUnicodeObject struct and another for its data149 buffer. */150 unsigned int compact:1;151 /* The string only contains characters in the range U+0000-U+007F (ASCII)152 and the kind is PyUnicode_1BYTE_KIND. If ascii is set and compact is153 set, use the PyASCIIObject structure. */154 unsigned int ascii:1;155 /* The object is statically allocated. */156 unsigned int statically_allocated:1;157#ifndef Py_GIL_DISABLED158 /* Padding to ensure that PyUnicode_DATA() is always aligned to159 4 bytes (see issue gh-63736 on m68k) */160 unsigned int :24;161#endif162 } state;163} PyASCIIObject;164 165/* Non-ASCII strings allocated through PyUnicode_New use the166 PyCompactUnicodeObject structure. state.compact is set, and the data167 immediately follow the structure. */168typedef struct {169 PyASCIIObject _base;170 Py_ssize_t utf8_length; /* Number of bytes in utf8, excluding the171 * terminating \0. */172 char *utf8; /* UTF-8 representation (null-terminated) */173} PyCompactUnicodeObject;174 175/* Object format for Unicode subclasses. */176typedef struct {177 PyCompactUnicodeObject _base;178 union {179 void *any;180 Py_UCS1 *latin1;181 Py_UCS2 *ucs2;182 Py_UCS4 *ucs4;183 } data; /* Canonical, smallest-form Unicode buffer */184} PyUnicodeObject;185 186 187#define _PyASCIIObject_CAST(op) \188 (assert(PyUnicode_Check(op)), \189 _Py_CAST(PyASCIIObject*, (op)))190#define _PyCompactUnicodeObject_CAST(op) \191 (assert(PyUnicode_Check(op)), \192 _Py_CAST(PyCompactUnicodeObject*, (op)))193#define _PyUnicodeObject_CAST(op) \194 (assert(PyUnicode_Check(op)), \195 _Py_CAST(PyUnicodeObject*, (op)))196 197 198/* --- Flexible String Representation Helper Macros (PEP 393) -------------- */199 200/* Values for PyASCIIObject.state: */201 202/* Interning state. */203#define SSTATE_NOT_INTERNED 0204#define SSTATE_INTERNED_MORTAL 1205#define SSTATE_INTERNED_IMMORTAL 2206#define SSTATE_INTERNED_IMMORTAL_STATIC 3207 208/* Use only if you know it's a string */209static inline unsigned int PyUnicode_CHECK_INTERNED(PyObject *op) {210#ifdef Py_GIL_DISABLED211 return _Py_atomic_load_uint8_relaxed(&_PyASCIIObject_CAST(op)->state.interned);212#else213 return _PyASCIIObject_CAST(op)->state.interned;214#endif215}216#define PyUnicode_CHECK_INTERNED(op) PyUnicode_CHECK_INTERNED(_PyObject_CAST(op))217 218/* For backward compatibility. Soft-deprecated. */219static inline unsigned int PyUnicode_IS_READY(PyObject* Py_UNUSED(op)) {220 return 1;221}222#define PyUnicode_IS_READY(op) PyUnicode_IS_READY(_PyObject_CAST(op))223 224/* Return true if the string contains only ASCII characters, or 0 if not. The225 string may be compact (PyUnicode_IS_COMPACT_ASCII) or not. */226static inline unsigned int PyUnicode_IS_ASCII(PyObject *op) {227 return _PyASCIIObject_CAST(op)->state.ascii;228}229#define PyUnicode_IS_ASCII(op) PyUnicode_IS_ASCII(_PyObject_CAST(op))230 231/* Return true if the string is compact or 0 if not.232 No type checks are performed. */233static inline unsigned int PyUnicode_IS_COMPACT(PyObject *op) {234 return _PyASCIIObject_CAST(op)->state.compact;235}236#define PyUnicode_IS_COMPACT(op) PyUnicode_IS_COMPACT(_PyObject_CAST(op))237 238/* Return true if the string is a compact ASCII string (use PyASCIIObject239 structure), or 0 if not. No type checks are performed. */240static inline int PyUnicode_IS_COMPACT_ASCII(PyObject *op) {241 return (_PyASCIIObject_CAST(op)->state.ascii && PyUnicode_IS_COMPACT(op));242}243#define PyUnicode_IS_COMPACT_ASCII(op) PyUnicode_IS_COMPACT_ASCII(_PyObject_CAST(op))244 245enum PyUnicode_Kind {246/* Return values of the PyUnicode_KIND() function: */247 PyUnicode_1BYTE_KIND = 1,248 PyUnicode_2BYTE_KIND = 2,249 PyUnicode_4BYTE_KIND = 4250};251 252PyAPI_FUNC(int) PyUnicode_KIND(PyObject *op);253 254// PyUnicode_KIND(): Return one of the PyUnicode_*_KIND values defined above.255//256// gh-89653: Converting this macro to a static inline function would introduce257// new compiler warnings on "kind < PyUnicode_KIND(str)" (compare signed and258// unsigned numbers) where kind type is an int or on259// "unsigned int kind = PyUnicode_KIND(str)" (cast signed to unsigned).260#define PyUnicode_KIND(op) _Py_RVALUE(_PyASCIIObject_CAST(op)->state.kind)261 262/* Return a void pointer to the raw unicode buffer. */263static inline void* _PyUnicode_COMPACT_DATA(PyObject *op) {264 if (PyUnicode_IS_ASCII(op)) {265 return _Py_STATIC_CAST(void*, (_PyASCIIObject_CAST(op) + 1));266 }267 return _Py_STATIC_CAST(void*, (_PyCompactUnicodeObject_CAST(op) + 1));268}269 270static inline void* _PyUnicode_NONCOMPACT_DATA(PyObject *op) {271 void *data;272 assert(!PyUnicode_IS_COMPACT(op));273 data = _PyUnicodeObject_CAST(op)->data.any;274 assert(data != NULL);275 return data;276}277 278PyAPI_FUNC(void*) PyUnicode_DATA(PyObject *op);279 280static inline void* _PyUnicode_DATA(PyObject *op) {281 if (PyUnicode_IS_COMPACT(op)) {282 return _PyUnicode_COMPACT_DATA(op);283 }284 return _PyUnicode_NONCOMPACT_DATA(op);285}286#define PyUnicode_DATA(op) _PyUnicode_DATA(_PyObject_CAST(op))287 288/* Return pointers to the canonical representation cast to unsigned char,289 Py_UCS2, or Py_UCS4 for direct character access.290 No checks are performed, use PyUnicode_KIND() before to ensure291 these will work correctly. */292 293#define PyUnicode_1BYTE_DATA(op) _Py_STATIC_CAST(Py_UCS1*, PyUnicode_DATA(op))294#define PyUnicode_2BYTE_DATA(op) _Py_STATIC_CAST(Py_UCS2*, PyUnicode_DATA(op))295#define PyUnicode_4BYTE_DATA(op) _Py_STATIC_CAST(Py_UCS4*, PyUnicode_DATA(op))296 297/* Returns the length of the unicode string. */298static inline Py_ssize_t PyUnicode_GET_LENGTH(PyObject *op) {299 return _PyASCIIObject_CAST(op)->length;300}301#define PyUnicode_GET_LENGTH(op) PyUnicode_GET_LENGTH(_PyObject_CAST(op))302 303/* Write into the canonical representation, this function does not do any sanity304 checks and is intended for usage in loops. The caller should cache the305 kind and data pointers obtained from other function calls.306 index is the index in the string (starts at 0) and value is the new307 code point value which should be written to that location. */308static inline void PyUnicode_WRITE(int kind, void *data,309 Py_ssize_t index, Py_UCS4 value)310{311 assert(index >= 0);312 if (kind == PyUnicode_1BYTE_KIND) {313 assert(value <= 0xffU);314 _Py_STATIC_CAST(Py_UCS1*, data)[index] = _Py_STATIC_CAST(Py_UCS1, value);315 }316 else if (kind == PyUnicode_2BYTE_KIND) {317 assert(value <= 0xffffU);318 _Py_STATIC_CAST(Py_UCS2*, data)[index] = _Py_STATIC_CAST(Py_UCS2, value);319 }320 else {321 assert(kind == PyUnicode_4BYTE_KIND);322 assert(value <= 0x10ffffU);323 _Py_STATIC_CAST(Py_UCS4*, data)[index] = value;324 }325}326#define PyUnicode_WRITE(kind, data, index, value) \327 PyUnicode_WRITE(_Py_STATIC_CAST(int, kind), _Py_CAST(void*, data), \328 (index), _Py_STATIC_CAST(Py_UCS4, value))329 330/* Read a code point from the string's canonical representation. No checks331 are performed. */332static inline Py_UCS4 PyUnicode_READ(int kind,333 const void *data, Py_ssize_t index)334{335 assert(index >= 0);336 if (kind == PyUnicode_1BYTE_KIND) {337 return _Py_STATIC_CAST(const Py_UCS1*, data)[index];338 }339 if (kind == PyUnicode_2BYTE_KIND) {340 return _Py_STATIC_CAST(const Py_UCS2*, data)[index];341 }342 assert(kind == PyUnicode_4BYTE_KIND);343 return _Py_STATIC_CAST(const Py_UCS4*, data)[index];344}345#define PyUnicode_READ(kind, data, index) \346 PyUnicode_READ(_Py_STATIC_CAST(int, kind), \347 _Py_STATIC_CAST(const void*, data), \348 (index))349 350/* PyUnicode_READ_CHAR() is less efficient than PyUnicode_READ() because it351 calls PyUnicode_KIND() and might call it twice. For single reads, use352 PyUnicode_READ_CHAR, for multiple consecutive reads callers should353 cache kind and use PyUnicode_READ instead. */354static inline Py_UCS4 PyUnicode_READ_CHAR(PyObject *unicode, Py_ssize_t index)355{356 int kind;357 358 assert(index >= 0);359 // Tolerate reading the NUL character at str[len(str)]360 assert(index <= PyUnicode_GET_LENGTH(unicode));361 362 kind = PyUnicode_KIND(unicode);363 if (kind == PyUnicode_1BYTE_KIND) {364 return PyUnicode_1BYTE_DATA(unicode)[index];365 }366 if (kind == PyUnicode_2BYTE_KIND) {367 return PyUnicode_2BYTE_DATA(unicode)[index];368 }369 assert(kind == PyUnicode_4BYTE_KIND);370 return PyUnicode_4BYTE_DATA(unicode)[index];371}372#define PyUnicode_READ_CHAR(unicode, index) \373 PyUnicode_READ_CHAR(_PyObject_CAST(unicode), (index))374 375/* Return a maximum character value which is suitable for creating another376 string based on op. This is always an approximation but more efficient377 than iterating over the string. */378static inline Py_UCS4 PyUnicode_MAX_CHAR_VALUE(PyObject *op)379{380 int kind;381 382 if (PyUnicode_IS_ASCII(op)) {383 return 0x7fU;384 }385 386 kind = PyUnicode_KIND(op);387 if (kind == PyUnicode_1BYTE_KIND) {388 return 0xffU;389 }390 if (kind == PyUnicode_2BYTE_KIND) {391 return 0xffffU;392 }393 assert(kind == PyUnicode_4BYTE_KIND);394 return 0x10ffffU;395}396#define PyUnicode_MAX_CHAR_VALUE(op) \397 PyUnicode_MAX_CHAR_VALUE(_PyObject_CAST(op))398 399 400/* === Public API ========================================================= */401 402/* With PEP 393, this is the recommended way to allocate a new unicode object.403 This function will allocate the object and its buffer in a single memory404 block. Objects created using this function are not resizable. */405PyAPI_FUNC(PyObject*) PyUnicode_New(406 Py_ssize_t size, /* Number of code points in the new string */407 Py_UCS4 maxchar /* maximum code point value in the string */408 );409 410/* For backward compatibility. Soft-deprecated. */411static inline int PyUnicode_READY(PyObject* Py_UNUSED(op))412{413 return 0;414}415#define PyUnicode_READY(op) PyUnicode_READY(_PyObject_CAST(op))416 417/* Copy character from one unicode object into another, this function performs418 character conversion when necessary and falls back to memcpy() if possible.419 420 Fail if to is too small (smaller than *how_many* or smaller than421 len(from)-from_start), or if kind(from[from_start:from_start+how_many]) >422 kind(to), or if *to* has more than 1 reference.423 424 Return the number of written character, or return -1 and raise an exception425 on error.426 427 Pseudo-code:428 429 how_many = min(how_many, len(from) - from_start)430 to[to_start:to_start+how_many] = from[from_start:from_start+how_many]431 return how_many432 433 Note: The function doesn't write a terminating null character.434 */435PyAPI_FUNC(Py_ssize_t) PyUnicode_CopyCharacters(436 PyObject *to,437 Py_ssize_t to_start,438 PyObject *from,439 Py_ssize_t from_start,440 Py_ssize_t how_many441 );442 443/* Fill a string with a character: write fill_char into444 unicode[start:start+length].445 446 Fail if fill_char is bigger than the string maximum character, or if the447 string has more than 1 reference.448 449 Return the number of written character, or return -1 and raise an exception450 on error. */451PyAPI_FUNC(Py_ssize_t) PyUnicode_Fill(452 PyObject *unicode,453 Py_ssize_t start,454 Py_ssize_t length,455 Py_UCS4 fill_char456 );457 458/* Create a new string from a buffer of Py_UCS1, Py_UCS2 or Py_UCS4 characters.459 Scan the string to find the maximum character. */460PyAPI_FUNC(PyObject*) PyUnicode_FromKindAndData(461 int kind,462 const void *buffer,463 Py_ssize_t size);464 465 466/* --- Public PyUnicodeWriter API ----------------------------------------- */467 468typedef struct PyUnicodeWriter PyUnicodeWriter;469 470PyAPI_FUNC(PyUnicodeWriter*) PyUnicodeWriter_Create(Py_ssize_t length);471PyAPI_FUNC(void) PyUnicodeWriter_Discard(PyUnicodeWriter *writer);472PyAPI_FUNC(PyObject*) PyUnicodeWriter_Finish(PyUnicodeWriter *writer);473 474PyAPI_FUNC(int) PyUnicodeWriter_WriteChar(475 PyUnicodeWriter *writer,476 Py_UCS4 ch);477PyAPI_FUNC(int) PyUnicodeWriter_WriteUTF8(478 PyUnicodeWriter *writer,479 const char *str,480 Py_ssize_t size);481PyAPI_FUNC(int) PyUnicodeWriter_WriteASCII(482 PyUnicodeWriter *writer,483 const char *str,484 Py_ssize_t size);485PyAPI_FUNC(int) PyUnicodeWriter_WriteWideChar(486 PyUnicodeWriter *writer,487 const wchar_t *str,488 Py_ssize_t size);489PyAPI_FUNC(int) PyUnicodeWriter_WriteUCS4(490 PyUnicodeWriter *writer,491 Py_UCS4 *str,492 Py_ssize_t size);493 494PyAPI_FUNC(int) PyUnicodeWriter_WriteStr(495 PyUnicodeWriter *writer,496 PyObject *obj);497PyAPI_FUNC(int) PyUnicodeWriter_WriteRepr(498 PyUnicodeWriter *writer,499 PyObject *obj);500PyAPI_FUNC(int) PyUnicodeWriter_WriteSubstring(501 PyUnicodeWriter *writer,502 PyObject *str,503 Py_ssize_t start,504 Py_ssize_t end);505PyAPI_FUNC(int) PyUnicodeWriter_Format(506 PyUnicodeWriter *writer,507 const char *format,508 ...);509PyAPI_FUNC(int) PyUnicodeWriter_DecodeUTF8Stateful(510 PyUnicodeWriter *writer,511 const char *string, /* UTF-8 encoded string */512 Py_ssize_t length, /* size of string */513 const char *errors, /* error handling */514 Py_ssize_t *consumed); /* bytes consumed */515 516 517/* --- Private _PyUnicodeWriter API --------------------------------------- */518 519typedef struct {520 PyObject *buffer;521 void *data;522 int kind;523 Py_UCS4 maxchar;524 Py_ssize_t size;525 Py_ssize_t pos;526 527 /* minimum number of allocated characters (default: 0) */528 Py_ssize_t min_length;529 530 /* minimum character (default: 127, ASCII) */531 Py_UCS4 min_char;532 533 /* If non-zero, overallocate the buffer (default: 0). */534 unsigned char overallocate;535 536 /* If readonly is 1, buffer is a shared string (cannot be modified)537 and size is set to 0. */538 unsigned char readonly;539} _PyUnicodeWriter;540 541// Initialize a Unicode writer.542//543// By default, the minimum buffer size is 0 character and overallocation is544// disabled. Set min_length, min_char and overallocate attributes to control545// the allocation of the buffer.546_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(void) _PyUnicodeWriter_Init(547 _PyUnicodeWriter *writer);548 549/* Prepare the buffer to write 'length' characters550 with the specified maximum character.551 552 Return 0 on success, raise an exception and return -1 on error. */553#define _PyUnicodeWriter_Prepare(WRITER, LENGTH, MAXCHAR) \554 (((MAXCHAR) <= (WRITER)->maxchar \555 && (LENGTH) <= (WRITER)->size - (WRITER)->pos) \556 ? 0 \557 : (((LENGTH) == 0) \558 ? 0 \559 : _PyUnicodeWriter_PrepareInternal((WRITER), (LENGTH), (MAXCHAR))))560 561/* Don't call this function directly, use the _PyUnicodeWriter_Prepare() macro562 instead. */563_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_PrepareInternal(564 _PyUnicodeWriter *writer,565 Py_ssize_t length,566 Py_UCS4 maxchar);567 568/* Prepare the buffer to have at least the kind KIND.569 For example, kind=PyUnicode_2BYTE_KIND ensures that the writer will570 support characters in range U+000-U+FFFF.571 572 Return 0 on success, raise an exception and return -1 on error. */573#define _PyUnicodeWriter_PrepareKind(WRITER, KIND) \574 ((KIND) <= (WRITER)->kind \575 ? 0 \576 : _PyUnicodeWriter_PrepareKindInternal((WRITER), (KIND)))577 578/* Don't call this function directly, use the _PyUnicodeWriter_PrepareKind()579 macro instead. */580_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_PrepareKindInternal(581 _PyUnicodeWriter *writer,582 int kind);583 584/* Append a Unicode character.585 Return 0 on success, raise an exception and return -1 on error. */586_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_WriteChar(587 _PyUnicodeWriter *writer,588 Py_UCS4 ch);589 590/* Append a Unicode string.591 Return 0 on success, raise an exception and return -1 on error. */592_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_WriteStr(593 _PyUnicodeWriter *writer,594 PyObject *str); /* Unicode string */595 596/* Append a substring of a Unicode string.597 Return 0 on success, raise an exception and return -1 on error. */598_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_WriteSubstring(599 _PyUnicodeWriter *writer,600 PyObject *str, /* Unicode string */601 Py_ssize_t start,602 Py_ssize_t end);603 604/* Append an ASCII-encoded byte string.605 Return 0 on success, raise an exception and return -1 on error. */606_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_WriteASCIIString(607 _PyUnicodeWriter *writer,608 const char *str, /* ASCII-encoded byte string */609 Py_ssize_t len); /* number of bytes, or -1 if unknown */610 611/* Append a latin1-encoded byte string.612 Return 0 on success, raise an exception and return -1 on error. */613_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_WriteLatin1String(614 _PyUnicodeWriter *writer,615 const char *str, /* latin1-encoded byte string */616 Py_ssize_t len); /* length in bytes */617 618/* Get the value of the writer as a Unicode string. Clear the619 buffer of the writer. Raise an exception and return NULL620 on error. */621_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(PyObject *) _PyUnicodeWriter_Finish(622 _PyUnicodeWriter *writer);623 624/* Deallocate memory of a writer (clear its internal buffer). */625_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(void) _PyUnicodeWriter_Dealloc(626 _PyUnicodeWriter *writer);627 628 629/* --- Manage the default encoding ---------------------------------------- */630 631/* Returns a pointer to the default encoding (UTF-8) of the632 Unicode object unicode.633 634 Like PyUnicode_AsUTF8AndSize(), this also caches the UTF-8 representation635 in the unicodeobject.636 637 _PyUnicode_AsString is a #define for PyUnicode_AsUTF8 to638 support the previous internal function with the same behaviour.639 640 Use of this API is DEPRECATED since no size information can be641 extracted from the returned data.642*/643 644PyAPI_FUNC(const char *) PyUnicode_AsUTF8(PyObject *unicode);645 646// Deprecated alias kept for backward compatibility647Py_DEPRECATED(3.14) static inline const char*648_PyUnicode_AsString(PyObject *unicode)649{650 return PyUnicode_AsUTF8(unicode);651}652 653 654/* === Characters Type APIs =============================================== */655 656/* These should not be used directly. Use the Py_UNICODE_IS* and657 Py_UNICODE_TO* macros instead.658 659 These APIs are implemented in Objects/unicodectype.c.660 661*/662 663PyAPI_FUNC(int) _PyUnicode_IsLowercase(664 Py_UCS4 ch /* Unicode character */665 );666 667PyAPI_FUNC(int) _PyUnicode_IsUppercase(668 Py_UCS4 ch /* Unicode character */669 );670 671PyAPI_FUNC(int) _PyUnicode_IsTitlecase(672 Py_UCS4 ch /* Unicode character */673 );674 675PyAPI_FUNC(int) _PyUnicode_IsWhitespace(676 const Py_UCS4 ch /* Unicode character */677 );678 679PyAPI_FUNC(int) _PyUnicode_IsLinebreak(680 const Py_UCS4 ch /* Unicode character */681 );682 683PyAPI_FUNC(Py_UCS4) _PyUnicode_ToLowercase(684 Py_UCS4 ch /* Unicode character */685 );686 687PyAPI_FUNC(Py_UCS4) _PyUnicode_ToUppercase(688 Py_UCS4 ch /* Unicode character */689 );690 691PyAPI_FUNC(Py_UCS4) _PyUnicode_ToTitlecase(692 Py_UCS4 ch /* Unicode character */693 );694 695PyAPI_FUNC(int) _PyUnicode_ToDecimalDigit(696 Py_UCS4 ch /* Unicode character */697 );698 699PyAPI_FUNC(int) _PyUnicode_ToDigit(700 Py_UCS4 ch /* Unicode character */701 );702 703PyAPI_FUNC(double) _PyUnicode_ToNumeric(704 Py_UCS4 ch /* Unicode character */705 );706 707PyAPI_FUNC(int) _PyUnicode_IsDecimalDigit(708 Py_UCS4 ch /* Unicode character */709 );710 711PyAPI_FUNC(int) _PyUnicode_IsDigit(712 Py_UCS4 ch /* Unicode character */713 );714 715PyAPI_FUNC(int) _PyUnicode_IsNumeric(716 Py_UCS4 ch /* Unicode character */717 );718 719PyAPI_FUNC(int) _PyUnicode_IsPrintable(720 Py_UCS4 ch /* Unicode character */721 );722 723PyAPI_FUNC(int) _PyUnicode_IsAlpha(724 Py_UCS4 ch /* Unicode character */725 );726 727// Helper array used by Py_UNICODE_ISSPACE().728PyAPI_DATA(const unsigned char) _Py_ascii_whitespace[];729 730// Since splitting on whitespace is an important use case, and731// whitespace in most situations is solely ASCII whitespace, we732// optimize for the common case by using a quick look-up table733// _Py_ascii_whitespace (see below) with an inlined check.734static inline int Py_UNICODE_ISSPACE(Py_UCS4 ch) {735 if (ch < 128) {736 return _Py_ascii_whitespace[ch];737 }738 return _PyUnicode_IsWhitespace(ch);739}740 741#define Py_UNICODE_ISLOWER(ch) _PyUnicode_IsLowercase(ch)742#define Py_UNICODE_ISUPPER(ch) _PyUnicode_IsUppercase(ch)743#define Py_UNICODE_ISTITLE(ch) _PyUnicode_IsTitlecase(ch)744#define Py_UNICODE_ISLINEBREAK(ch) _PyUnicode_IsLinebreak(ch)745 746#define Py_UNICODE_TOLOWER(ch) _PyUnicode_ToLowercase(ch)747#define Py_UNICODE_TOUPPER(ch) _PyUnicode_ToUppercase(ch)748#define Py_UNICODE_TOTITLE(ch) _PyUnicode_ToTitlecase(ch)749 750#define Py_UNICODE_ISDECIMAL(ch) _PyUnicode_IsDecimalDigit(ch)751#define Py_UNICODE_ISDIGIT(ch) _PyUnicode_IsDigit(ch)752#define Py_UNICODE_ISNUMERIC(ch) _PyUnicode_IsNumeric(ch)753#define Py_UNICODE_ISPRINTABLE(ch) _PyUnicode_IsPrintable(ch)754 755#define Py_UNICODE_TODECIMAL(ch) _PyUnicode_ToDecimalDigit(ch)756#define Py_UNICODE_TODIGIT(ch) _PyUnicode_ToDigit(ch)757#define Py_UNICODE_TONUMERIC(ch) _PyUnicode_ToNumeric(ch)758 759#define Py_UNICODE_ISALPHA(ch) _PyUnicode_IsAlpha(ch)760 761static inline int Py_UNICODE_ISALNUM(Py_UCS4 ch) {762 return (Py_UNICODE_ISALPHA(ch)763 || Py_UNICODE_ISDECIMAL(ch)764 || Py_UNICODE_ISDIGIT(ch)765 || Py_UNICODE_ISNUMERIC(ch));766}767 768 769/* === Misc functions ===================================================== */770 771// Return an interned Unicode object for an Identifier; may fail if there is no772// memory.773PyAPI_FUNC(PyObject*) _PyUnicode_FromId(_Py_Identifier*);774 