codekingpro/portable-devtools
115k
1#ifndef Py_UNICODEOBJECT_H2#define Py_UNICODEOBJECT_H3 4/*5 6Unicode implementation based on original code by Fredrik Lundh,7modified by Marc-Andre Lemburg (mal@lemburg.com) according to the8Unicode Integration Proposal. (See9http://www.egenix.com/files/python/unicode-proposal.txt).10 11Copyright (c) Corporation for National Research Initiatives.12 13 14 Original header:15 --------------------------------------------------------------------16 17 * Yet another Unicode string type for Python. This type supports the18 * 16-bit Basic Multilingual Plane (BMP) only.19 *20 * Written by Fredrik Lundh, January 1999.21 *22 * Copyright (c) 1999 by Secret Labs AB.23 * Copyright (c) 1999 by Fredrik Lundh.24 *25 * fredrik@pythonware.com26 * http://www.pythonware.com27 *28 * --------------------------------------------------------------------29 * This Unicode String Type is30 *31 * Copyright (c) 1999 by Secret Labs AB32 * Copyright (c) 1999 by Fredrik Lundh33 *34 * By obtaining, using, and/or copying this software and/or its35 * associated documentation, you agree that you have read, understood,36 * and will comply with the following terms and conditions:37 *38 * Permission to use, copy, modify, and distribute this software and its39 * associated documentation for any purpose and without fee is hereby40 * granted, provided that the above copyright notice appears in all41 * copies, and that both that copyright notice and this permission notice42 * appear in supporting documentation, and that the name of Secret Labs43 * AB or the author not be used in advertising or publicity pertaining to44 * distribution of the software without specific, written prior45 * permission.46 *47 * SECRET LABS AB AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH REGARD TO48 * THIS SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND49 * FITNESS. IN NO EVENT SHALL SECRET LABS AB OR THE AUTHOR BE LIABLE FOR50 * ANY SPECIAL, INDIRECT OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES51 * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN52 * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT53 * OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.54 * -------------------------------------------------------------------- */55 56/* === Internal API ======================================================= */57 58/* --- Internal Unicode Format -------------------------------------------- */59 60/* Python 3.x requires unicode */61#define Py_USING_UNICODE62 63#ifndef SIZEOF_WCHAR_T64#error Must define SIZEOF_WCHAR_T65#endif66 67#define Py_UNICODE_SIZE SIZEOF_WCHAR_T68 69/* If wchar_t can be used for UCS-4 storage, set Py_UNICODE_WIDE.70 Otherwise, Unicode strings are stored as UCS-2 (with limited support71 for UTF-16) */72 73#if Py_UNICODE_SIZE >= 474#define Py_UNICODE_WIDE75#endif76 77/* Set these flags if the platform has "wchar.h" and the78 wchar_t type is a 16-bit unsigned type */79/* #define HAVE_WCHAR_H */80/* #define HAVE_USABLE_WCHAR_T */81 82/* If the compiler provides a wchar_t type we try to support it83 through the interface functions PyUnicode_FromWideChar(),84 PyUnicode_AsWideChar() and PyUnicode_AsWideCharString(). */85 86#ifdef HAVE_USABLE_WCHAR_T87# ifndef HAVE_WCHAR_H88# define HAVE_WCHAR_H89# endif90#endif91 92/* Py_UCS4 and Py_UCS2 are typedefs for the respective93 unicode representations. */94typedef uint32_t Py_UCS4;95typedef uint16_t Py_UCS2;96typedef uint8_t Py_UCS1;97 98#ifdef __cplusplus99extern "C" {100#endif101 102 103PyAPI_DATA(PyTypeObject) PyUnicode_Type;104PyAPI_DATA(PyTypeObject) PyUnicodeIter_Type;105 106#define PyUnicode_Check(op) \107 PyType_FastSubclass(Py_TYPE(op), Py_TPFLAGS_UNICODE_SUBCLASS)108#define PyUnicode_CheckExact(op) Py_IS_TYPE((op), &PyUnicode_Type)109 110/* --- Constants ---------------------------------------------------------- */111 112/* This Unicode character will be used as replacement character during113 decoding if the errors argument is set to "replace". Note: the114 Unicode character U+FFFD is the official REPLACEMENT CHARACTER in115 Unicode 3.0. */116 117#define Py_UNICODE_REPLACEMENT_CHARACTER ((Py_UCS4) 0xFFFD)118 119/* === Public API ========================================================= */120 121/* Similar to PyUnicode_FromUnicode(), but u points to UTF-8 encoded bytes */122PyAPI_FUNC(PyObject*) PyUnicode_FromStringAndSize(123 const char *u, /* UTF-8 encoded string */124 Py_ssize_t size /* size of buffer */125 );126 127/* Similar to PyUnicode_FromUnicode(), but u points to null-terminated128 UTF-8 encoded bytes. The size is determined with strlen(). */129PyAPI_FUNC(PyObject*) PyUnicode_FromString(130 const char *u /* UTF-8 encoded string */131 );132 133#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000134PyAPI_FUNC(PyObject*) PyUnicode_Substring(135 PyObject *str,136 Py_ssize_t start,137 Py_ssize_t end);138#endif139 140#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000141/* Copy the string into a UCS4 buffer including the null character if copy_null142 is set. Return NULL and raise an exception on error. Raise a SystemError if143 the buffer is smaller than the string. Return buffer on success.144 145 buflen is the length of the buffer in (Py_UCS4) characters. */146PyAPI_FUNC(Py_UCS4*) PyUnicode_AsUCS4(147 PyObject *unicode,148 Py_UCS4* buffer,149 Py_ssize_t buflen,150 int copy_null);151 152/* Copy the string into a UCS4 buffer. A new buffer is allocated using153 * PyMem_Malloc; if this fails, NULL is returned with a memory error154 exception set. */155PyAPI_FUNC(Py_UCS4*) PyUnicode_AsUCS4Copy(PyObject *unicode);156#endif157 158#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000159/* Get the length of the Unicode object. */160 161PyAPI_FUNC(Py_ssize_t) PyUnicode_GetLength(162 PyObject *unicode163);164#endif165 166#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000167/* Read a character from the string. */168 169PyAPI_FUNC(Py_UCS4) PyUnicode_ReadChar(170 PyObject *unicode,171 Py_ssize_t index172 );173 174/* Write a character to the string. The string must have been created through175 PyUnicode_New, must not be shared, and must not have been hashed yet.176 177 Return 0 on success, -1 on error. */178 179PyAPI_FUNC(int) PyUnicode_WriteChar(180 PyObject *unicode,181 Py_ssize_t index,182 Py_UCS4 character183 );184#endif185 186/* Resize a Unicode object. The length is the number of codepoints.187 188 *unicode is modified to point to the new (resized) object and 0189 returned on success.190 191 Try to resize the string in place (which is usually faster than allocating192 a new string and copy characters), or create a new string.193 194 Error handling is implemented as follows: an exception is set, -1195 is returned and *unicode left untouched.196 197 WARNING: The function doesn't check string content, the result may not be a198 string in canonical representation. */199 200PyAPI_FUNC(int) PyUnicode_Resize(201 PyObject **unicode, /* Pointer to the Unicode object */202 Py_ssize_t length /* New length */203 );204 205/* Decode obj to a Unicode object.206 207 bytes, bytearray and other bytes-like objects are decoded according to the208 given encoding and error handler. The encoding and error handler can be209 NULL to have the interface use UTF-8 and "strict".210 211 All other objects (including Unicode objects) raise an exception.212 213 The API returns NULL in case of an error. The caller is responsible214 for decref'ing the returned objects.215 216*/217 218PyAPI_FUNC(PyObject*) PyUnicode_FromEncodedObject(219 PyObject *obj, /* Object */220 const char *encoding, /* encoding */221 const char *errors /* error handling */222 );223 224/* Copy an instance of a Unicode subtype to a new true Unicode object if225 necessary. If obj is already a true Unicode object (not a subtype), return226 the reference with *incremented* refcount.227 228 The API returns NULL in case of an error. The caller is responsible229 for decref'ing the returned objects.230 231*/232 233PyAPI_FUNC(PyObject*) PyUnicode_FromObject(234 PyObject *obj /* Object */235 );236 237PyAPI_FUNC(PyObject *) PyUnicode_FromFormatV(238 const char *format, /* ASCII-encoded string */239 va_list vargs240 );241PyAPI_FUNC(PyObject *) PyUnicode_FromFormat(242 const char *format, /* ASCII-encoded string */243 ...244 );245 246PyAPI_FUNC(void) PyUnicode_InternInPlace(PyObject **);247PyAPI_FUNC(PyObject *) PyUnicode_InternFromString(248 const char *u /* UTF-8 encoded string */249 );250 251/* --- wchar_t support for platforms which support it --------------------- */252 253#ifdef HAVE_WCHAR_H254 255/* Create a Unicode Object from the wchar_t buffer w of the given256 size.257 258 The buffer is copied into the new object. */259 260PyAPI_FUNC(PyObject*) PyUnicode_FromWideChar(261 const wchar_t *w, /* wchar_t buffer */262 Py_ssize_t size /* size of buffer */263 );264 265/* Copies the Unicode Object contents into the wchar_t buffer w. At266 most size wchar_t characters are copied.267 268 Note that the resulting wchar_t string may or may not be269 0-terminated. It is the responsibility of the caller to make sure270 that the wchar_t string is 0-terminated in case this is required by271 the application.272 273 Returns the number of wchar_t characters copied (excluding a274 possibly trailing 0-termination character) or -1 in case of an275 error. */276 277PyAPI_FUNC(Py_ssize_t) PyUnicode_AsWideChar(278 PyObject *unicode, /* Unicode object */279 wchar_t *w, /* wchar_t buffer */280 Py_ssize_t size /* size of buffer */281 );282 283/* Convert the Unicode object to a wide character string. The output string284 always ends with a nul character. If size is not NULL, write the number of285 wide characters (excluding the null character) into *size.286 287 Returns a buffer allocated by PyMem_Malloc() (use PyMem_Free() to free it)288 on success. On error, returns NULL, *size is undefined and raises a289 MemoryError. */290 291PyAPI_FUNC(wchar_t*) PyUnicode_AsWideCharString(292 PyObject *unicode, /* Unicode object */293 Py_ssize_t *size /* number of characters of the result */294 );295 296#endif297 298/* --- Unicode ordinals --------------------------------------------------- */299 300/* Create a Unicode Object from the given Unicode code point ordinal.301 302 The ordinal must be in range(0x110000). A ValueError is303 raised in case it is not.304 305*/306 307PyAPI_FUNC(PyObject*) PyUnicode_FromOrdinal(int ordinal);308 309/* === Builtin Codecs =====================================================310 311 Many of these APIs take two arguments encoding and errors. These312 parameters encoding and errors have the same semantics as the ones313 of the builtin str() API.314 315 Setting encoding to NULL causes the default encoding (UTF-8) to be used.316 317 Error handling is set by errors which may also be set to NULL318 meaning to use the default handling defined for the codec. Default319 error handling for all builtin codecs is "strict" (ValueErrors are320 raised).321 322 The codecs all use a similar interface. Only deviation from the323 generic ones are documented.324 325*/326 327/* --- Manage the default encoding ---------------------------------------- */328 329/* Returns "utf-8". */330PyAPI_FUNC(const char*) PyUnicode_GetDefaultEncoding(void);331 332/* --- Generic Codecs ----------------------------------------------------- */333 334/* Create a Unicode object by decoding the encoded string s of the335 given size. */336 337PyAPI_FUNC(PyObject*) PyUnicode_Decode(338 const char *s, /* encoded string */339 Py_ssize_t size, /* size of buffer */340 const char *encoding, /* encoding */341 const char *errors /* error handling */342 );343 344/* Decode a Unicode object unicode and return the result as Python345 object.346 347 This API is DEPRECATED and will be removed in 3.15.348 The only supported standard encoding is rot13.349 Use PyCodec_Decode() to decode with rot13 and non-standard codecs350 that decode from str. */351 352Py_DEPRECATED(3.6) PyAPI_FUNC(PyObject*) PyUnicode_AsDecodedObject(353 PyObject *unicode, /* Unicode object */354 const char *encoding, /* encoding */355 const char *errors /* error handling */356 );357 358/* Decode a Unicode object unicode and return the result as Unicode359 object.360 361 This API is DEPRECATED and will be removed in 3.15.362 The only supported standard encoding is rot13.363 Use PyCodec_Decode() to decode with rot13 and non-standard codecs364 that decode from str to str. */365 366Py_DEPRECATED(3.6) PyAPI_FUNC(PyObject*) PyUnicode_AsDecodedUnicode(367 PyObject *unicode, /* Unicode object */368 const char *encoding, /* encoding */369 const char *errors /* error handling */370 );371 372/* Encodes a Unicode object and returns the result as Python373 object.374 375 This API is DEPRECATED and will be removed in 3.15.376 It is superseded by PyUnicode_AsEncodedString()377 since all standard encodings (except rot13) encode str to bytes.378 Use PyCodec_Encode() for encoding with rot13 and non-standard codecs379 that encode form str to non-bytes. */380 381Py_DEPRECATED(3.6) PyAPI_FUNC(PyObject*) PyUnicode_AsEncodedObject(382 PyObject *unicode, /* Unicode object */383 const char *encoding, /* encoding */384 const char *errors /* error handling */385 );386 387/* Encodes a Unicode object and returns the result as Python string388 object. */389 390PyAPI_FUNC(PyObject*) PyUnicode_AsEncodedString(391 PyObject *unicode, /* Unicode object */392 const char *encoding, /* encoding */393 const char *errors /* error handling */394 );395 396/* Encodes a Unicode object and returns the result as Unicode397 object.398 399 This API is DEPRECATED and will be removed in 3.15.400 The only supported standard encodings is rot13.401 Use PyCodec_Encode() to encode with rot13 and non-standard codecs402 that encode from str to str. */403 404Py_DEPRECATED(3.6) PyAPI_FUNC(PyObject*) PyUnicode_AsEncodedUnicode(405 PyObject *unicode, /* Unicode object */406 const char *encoding, /* encoding */407 const char *errors /* error handling */408 );409 410/* Build an encoding map. */411 412PyAPI_FUNC(PyObject*) PyUnicode_BuildEncodingMap(413 PyObject* string /* 256 character map */414 );415 416/* --- UTF-7 Codecs ------------------------------------------------------- */417 418PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF7(419 const char *string, /* UTF-7 encoded string */420 Py_ssize_t length, /* size of string */421 const char *errors /* error handling */422 );423 424PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF7Stateful(425 const char *string, /* UTF-7 encoded string */426 Py_ssize_t length, /* size of string */427 const char *errors, /* error handling */428 Py_ssize_t *consumed /* bytes consumed */429 );430 431/* --- UTF-8 Codecs ------------------------------------------------------- */432 433PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF8(434 const char *string, /* UTF-8 encoded string */435 Py_ssize_t length, /* size of string */436 const char *errors /* error handling */437 );438 439PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF8Stateful(440 const char *string, /* UTF-8 encoded string */441 Py_ssize_t length, /* size of string */442 const char *errors, /* error handling */443 Py_ssize_t *consumed /* bytes consumed */444 );445 446PyAPI_FUNC(PyObject*) PyUnicode_AsUTF8String(447 PyObject *unicode /* Unicode object */448 );449 450/* Returns a pointer to the default encoding (UTF-8) of the451 Unicode object unicode and the size of the encoded representation452 in bytes stored in *size.453 454 In case of an error, no *size is set.455 456 This function caches the UTF-8 encoded string in the unicodeobject457 and subsequent calls will return the same string. The memory is released458 when the unicodeobject is deallocated.459*/460 461#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x030A0000462PyAPI_FUNC(const char *) PyUnicode_AsUTF8AndSize(463 PyObject *unicode,464 Py_ssize_t *size);465#endif466 467/* --- UTF-32 Codecs ------------------------------------------------------ */468 469/* Decodes length bytes from a UTF-32 encoded buffer string and returns470 the corresponding Unicode object.471 472 errors (if non-NULL) defines the error handling. It defaults473 to "strict".474 475 If byteorder is non-NULL, the decoder starts decoding using the476 given byte order:477 478 *byteorder == -1: little endian479 *byteorder == 0: native order480 *byteorder == 1: big endian481 482 In native mode, the first four bytes of the stream are checked for a483 BOM mark. If found, the BOM mark is analysed, the byte order484 adjusted and the BOM skipped. In the other modes, no BOM mark485 interpretation is done. After completion, *byteorder is set to the486 current byte order at the end of input data.487 488 If byteorder is NULL, the codec starts in native order mode.489 490*/491 492PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF32(493 const char *string, /* UTF-32 encoded string */494 Py_ssize_t length, /* size of string */495 const char *errors, /* error handling */496 int *byteorder /* pointer to byteorder to use497 0=native;-1=LE,1=BE; updated on498 exit */499 );500 501PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF32Stateful(502 const char *string, /* UTF-32 encoded string */503 Py_ssize_t length, /* size of string */504 const char *errors, /* error handling */505 int *byteorder, /* pointer to byteorder to use506 0=native;-1=LE,1=BE; updated on507 exit */508 Py_ssize_t *consumed /* bytes consumed */509 );510 511/* Returns a Python string using the UTF-32 encoding in native byte512 order. The string always starts with a BOM mark. */513 514PyAPI_FUNC(PyObject*) PyUnicode_AsUTF32String(515 PyObject *unicode /* Unicode object */516 );517 518/* Returns a Python string object holding the UTF-32 encoded value of519 the Unicode data.520 521 If byteorder is not 0, output is written according to the following522 byte order:523 524 byteorder == -1: little endian525 byteorder == 0: native byte order (writes a BOM mark)526 byteorder == 1: big endian527 528 If byteorder is 0, the output string will always start with the529 Unicode BOM mark (U+FEFF). In the other two modes, no BOM mark is530 prepended.531 532*/533 534/* --- UTF-16 Codecs ------------------------------------------------------ */535 536/* Decodes length bytes from a UTF-16 encoded buffer string and returns537 the corresponding Unicode object.538 539 errors (if non-NULL) defines the error handling. It defaults540 to "strict".541 542 If byteorder is non-NULL, the decoder starts decoding using the543 given byte order:544 545 *byteorder == -1: little endian546 *byteorder == 0: native order547 *byteorder == 1: big endian548 549 In native mode, the first two bytes of the stream are checked for a550 BOM mark. If found, the BOM mark is analysed, the byte order551 adjusted and the BOM skipped. In the other modes, no BOM mark552 interpretation is done. After completion, *byteorder is set to the553 current byte order at the end of input data.554 555 If byteorder is NULL, the codec starts in native order mode.556 557*/558 559PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF16(560 const char *string, /* UTF-16 encoded string */561 Py_ssize_t length, /* size of string */562 const char *errors, /* error handling */563 int *byteorder /* pointer to byteorder to use564 0=native;-1=LE,1=BE; updated on565 exit */566 );567 568PyAPI_FUNC(PyObject*) PyUnicode_DecodeUTF16Stateful(569 const char *string, /* UTF-16 encoded string */570 Py_ssize_t length, /* size of string */571 const char *errors, /* error handling */572 int *byteorder, /* pointer to byteorder to use573 0=native;-1=LE,1=BE; updated on574 exit */575 Py_ssize_t *consumed /* bytes consumed */576 );577 578/* Returns a Python string using the UTF-16 encoding in native byte579 order. The string always starts with a BOM mark. */580 581PyAPI_FUNC(PyObject*) PyUnicode_AsUTF16String(582 PyObject *unicode /* Unicode object */583 );584 585/* --- Unicode-Escape Codecs ---------------------------------------------- */586 587PyAPI_FUNC(PyObject*) PyUnicode_DecodeUnicodeEscape(588 const char *string, /* Unicode-Escape encoded string */589 Py_ssize_t length, /* size of string */590 const char *errors /* error handling */591 );592 593PyAPI_FUNC(PyObject*) PyUnicode_AsUnicodeEscapeString(594 PyObject *unicode /* Unicode object */595 );596 597/* --- Raw-Unicode-Escape Codecs ------------------------------------------ */598 599PyAPI_FUNC(PyObject*) PyUnicode_DecodeRawUnicodeEscape(600 const char *string, /* Raw-Unicode-Escape encoded string */601 Py_ssize_t length, /* size of string */602 const char *errors /* error handling */603 );604 605PyAPI_FUNC(PyObject*) PyUnicode_AsRawUnicodeEscapeString(606 PyObject *unicode /* Unicode object */607 );608 609/* --- Latin-1 Codecs -----------------------------------------------------610 611 Note: Latin-1 corresponds to the first 256 Unicode ordinals. */612 613PyAPI_FUNC(PyObject*) PyUnicode_DecodeLatin1(614 const char *string, /* Latin-1 encoded string */615 Py_ssize_t length, /* size of string */616 const char *errors /* error handling */617 );618 619PyAPI_FUNC(PyObject*) PyUnicode_AsLatin1String(620 PyObject *unicode /* Unicode object */621 );622 623/* --- ASCII Codecs -------------------------------------------------------624 625 Only 7-bit ASCII data is expected. All other codes generate errors.626 627*/628 629PyAPI_FUNC(PyObject*) PyUnicode_DecodeASCII(630 const char *string, /* ASCII encoded string */631 Py_ssize_t length, /* size of string */632 const char *errors /* error handling */633 );634 635PyAPI_FUNC(PyObject*) PyUnicode_AsASCIIString(636 PyObject *unicode /* Unicode object */637 );638 639/* --- Character Map Codecs -----------------------------------------------640 641 This codec uses mappings to encode and decode characters.642 643 Decoding mappings must map byte ordinals (integers in the range from 0 to644 255) to Unicode strings, integers (which are then interpreted as Unicode645 ordinals) or None. Unmapped data bytes (ones which cause a LookupError)646 as well as mapped to None, 0xFFFE or '\ufffe' are treated as "undefined647 mapping" and cause an error.648 649 Encoding mappings must map Unicode ordinal integers to bytes objects,650 integers in the range from 0 to 255 or None. Unmapped character651 ordinals (ones which cause a LookupError) as well as mapped to652 None are treated as "undefined mapping" and cause an error.653 654*/655 656PyAPI_FUNC(PyObject*) PyUnicode_DecodeCharmap(657 const char *string, /* Encoded string */658 Py_ssize_t length, /* size of string */659 PyObject *mapping, /* decoding mapping */660 const char *errors /* error handling */661 );662 663PyAPI_FUNC(PyObject*) PyUnicode_AsCharmapString(664 PyObject *unicode, /* Unicode object */665 PyObject *mapping /* encoding mapping */666 );667 668/* --- MBCS codecs for Windows -------------------------------------------- */669 670#ifdef MS_WINDOWS671PyAPI_FUNC(PyObject*) PyUnicode_DecodeMBCS(672 const char *string, /* MBCS encoded string */673 Py_ssize_t length, /* size of string */674 const char *errors /* error handling */675 );676 677PyAPI_FUNC(PyObject*) PyUnicode_DecodeMBCSStateful(678 const char *string, /* MBCS encoded string */679 Py_ssize_t length, /* size of string */680 const char *errors, /* error handling */681 Py_ssize_t *consumed /* bytes consumed */682 );683 684#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000685PyAPI_FUNC(PyObject*) PyUnicode_DecodeCodePageStateful(686 int code_page, /* code page number */687 const char *string, /* encoded string */688 Py_ssize_t length, /* size of string */689 const char *errors, /* error handling */690 Py_ssize_t *consumed /* bytes consumed */691 );692#endif693 694PyAPI_FUNC(PyObject*) PyUnicode_AsMBCSString(695 PyObject *unicode /* Unicode object */696 );697 698#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000699PyAPI_FUNC(PyObject*) PyUnicode_EncodeCodePage(700 int code_page, /* code page number */701 PyObject *unicode, /* Unicode object */702 const char *errors /* error handling */703 );704#endif705 706#endif /* MS_WINDOWS */707 708/* --- Locale encoding --------------------------------------------------- */709 710#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000711/* Decode a string from the current locale encoding. The decoder is strict if712 *surrogateescape* is equal to zero, otherwise it uses the 'surrogateescape'713 error handler (PEP 383) to escape undecodable bytes. If a byte sequence can714 be decoded as a surrogate character and *surrogateescape* is not equal to715 zero, the byte sequence is escaped using the 'surrogateescape' error handler716 instead of being decoded. *str* must end with a null character but cannot717 contain embedded null characters. */718 719PyAPI_FUNC(PyObject*) PyUnicode_DecodeLocaleAndSize(720 const char *str,721 Py_ssize_t len,722 const char *errors);723 724/* Similar to PyUnicode_DecodeLocaleAndSize(), but compute the string725 length using strlen(). */726 727PyAPI_FUNC(PyObject*) PyUnicode_DecodeLocale(728 const char *str,729 const char *errors);730 731/* Encode a Unicode object to the current locale encoding. The encoder is732 strict is *surrogateescape* is equal to zero, otherwise the733 "surrogateescape" error handler is used. Return a bytes object. The string734 cannot contain embedded null characters. */735 736PyAPI_FUNC(PyObject*) PyUnicode_EncodeLocale(737 PyObject *unicode,738 const char *errors739 );740#endif741 742/* --- File system encoding ---------------------------------------------- */743 744/* ParseTuple converter: encode str objects to bytes using745 PyUnicode_EncodeFSDefault(); bytes objects are output as-is. */746 747PyAPI_FUNC(int) PyUnicode_FSConverter(PyObject*, void*);748 749/* ParseTuple converter: decode bytes objects to unicode using750 PyUnicode_DecodeFSDefaultAndSize(); str objects are output as-is. */751 752PyAPI_FUNC(int) PyUnicode_FSDecoder(PyObject*, void*);753 754/* Decode a null-terminated string from the Python filesystem encoding755 and error handler.756 757 If the string length is known, use PyUnicode_DecodeFSDefaultAndSize(). */758PyAPI_FUNC(PyObject*) PyUnicode_DecodeFSDefault(759 const char *s /* encoded string */760 );761 762/* Decode a string from the Python filesystem encoding and error handler. */763PyAPI_FUNC(PyObject*) PyUnicode_DecodeFSDefaultAndSize(764 const char *s, /* encoded string */765 Py_ssize_t size /* size */766 );767 768/* Encode a Unicode object to the Python filesystem encoding and error handler.769 Return bytes. */770PyAPI_FUNC(PyObject*) PyUnicode_EncodeFSDefault(771 PyObject *unicode772 );773 774/* --- Methods & Slots ----------------------------------------------------775 776 These are capable of handling Unicode objects and strings on input777 (we refer to them as strings in the descriptions) and return778 Unicode objects or integers as appropriate. */779 780/* Concat two strings giving a new Unicode string. */781 782PyAPI_FUNC(PyObject*) PyUnicode_Concat(783 PyObject *left, /* Left string */784 PyObject *right /* Right string */785 );786 787/* Concat two strings and put the result in *pleft788 (sets *pleft to NULL on error) */789 790PyAPI_FUNC(void) PyUnicode_Append(791 PyObject **pleft, /* Pointer to left string */792 PyObject *right /* Right string */793 );794 795/* Concat two strings, put the result in *pleft and drop the right object796 (sets *pleft to NULL on error) */797 798PyAPI_FUNC(void) PyUnicode_AppendAndDel(799 PyObject **pleft, /* Pointer to left string */800 PyObject *right /* Right string */801 );802 803/* Split a string giving a list of Unicode strings.804 805 If sep is NULL, splitting will be done at all whitespace806 substrings. Otherwise, splits occur at the given separator.807 808 At most maxsplit splits will be done. If negative, no limit is set.809 810 Separators are not included in the resulting list.811 812*/813 814PyAPI_FUNC(PyObject*) PyUnicode_Split(815 PyObject *s, /* String to split */816 PyObject *sep, /* String separator */817 Py_ssize_t maxsplit /* Maxsplit count */818 );819 820/* Dito, but split at line breaks.821 822 CRLF is considered to be one line break. Line breaks are not823 included in the resulting list. */824 825PyAPI_FUNC(PyObject*) PyUnicode_Splitlines(826 PyObject *s, /* String to split */827 int keepends /* If true, line end markers are included */828 );829 830/* Partition a string using a given separator. */831 832PyAPI_FUNC(PyObject*) PyUnicode_Partition(833 PyObject *s, /* String to partition */834 PyObject *sep /* String separator */835 );836 837/* Partition a string using a given separator, searching from the end of the838 string. */839 840PyAPI_FUNC(PyObject*) PyUnicode_RPartition(841 PyObject *s, /* String to partition */842 PyObject *sep /* String separator */843 );844 845/* Split a string giving a list of Unicode strings.846 847 If sep is NULL, splitting will be done at all whitespace848 substrings. Otherwise, splits occur at the given separator.849 850 At most maxsplit splits will be done. But unlike PyUnicode_Split851 PyUnicode_RSplit splits from the end of the string. If negative,852 no limit is set.853 854 Separators are not included in the resulting list.855 856*/857 858PyAPI_FUNC(PyObject*) PyUnicode_RSplit(859 PyObject *s, /* String to split */860 PyObject *sep, /* String separator */861 Py_ssize_t maxsplit /* Maxsplit count */862 );863 864/* Translate a string by applying a character mapping table to it and865 return the resulting Unicode object.866 867 The mapping table must map Unicode ordinal integers to Unicode strings,868 Unicode ordinal integers or None (causing deletion of the character).869 870 Mapping tables may be dictionaries or sequences. Unmapped character871 ordinals (ones which cause a LookupError) are left untouched and872 are copied as-is.873 874*/875 876PyAPI_FUNC(PyObject *) PyUnicode_Translate(877 PyObject *str, /* String */878 PyObject *table, /* Translate table */879 const char *errors /* error handling */880 );881 882/* Join a sequence of strings using the given separator and return883 the resulting Unicode string. */884 885PyAPI_FUNC(PyObject*) PyUnicode_Join(886 PyObject *separator, /* Separator string */887 PyObject *seq /* Sequence object */888 );889 890/* Return 1 if substr matches str[start:end] at the given tail end, 0891 otherwise. */892 893PyAPI_FUNC(Py_ssize_t) PyUnicode_Tailmatch(894 PyObject *str, /* String */895 PyObject *substr, /* Prefix or Suffix string */896 Py_ssize_t start, /* Start index */897 Py_ssize_t end, /* Stop index */898 int direction /* Tail end: -1 prefix, +1 suffix */899 );900 901/* Return the first position of substr in str[start:end] using the902 given search direction or -1 if not found. -2 is returned in case903 an error occurred and an exception is set. */904 905PyAPI_FUNC(Py_ssize_t) PyUnicode_Find(906 PyObject *str, /* String */907 PyObject *substr, /* Substring to find */908 Py_ssize_t start, /* Start index */909 Py_ssize_t end, /* Stop index */910 int direction /* Find direction: +1 forward, -1 backward */911 );912 913#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x03030000914/* Like PyUnicode_Find, but search for single character only. */915PyAPI_FUNC(Py_ssize_t) PyUnicode_FindChar(916 PyObject *str,917 Py_UCS4 ch,918 Py_ssize_t start,919 Py_ssize_t end,920 int direction921 );922#endif923 924/* Count the number of occurrences of substr in str[start:end]. */925 926PyAPI_FUNC(Py_ssize_t) PyUnicode_Count(927 PyObject *str, /* String */928 PyObject *substr, /* Substring to count */929 Py_ssize_t start, /* Start index */930 Py_ssize_t end /* Stop index */931 );932 933/* Replace at most maxcount occurrences of substr in str with replstr934 and return the resulting Unicode object. */935 936PyAPI_FUNC(PyObject *) PyUnicode_Replace(937 PyObject *str, /* String */938 PyObject *substr, /* Substring to find */939 PyObject *replstr, /* Substring to replace */940 Py_ssize_t maxcount /* Max. number of replacements to apply;941 -1 = all */942 );943 944/* Compare two strings and return -1, 0, 1 for less than, equal,945 greater than resp.946 Raise an exception and return -1 on error. */947 948PyAPI_FUNC(int) PyUnicode_Compare(949 PyObject *left, /* Left string */950 PyObject *right /* Right string */951 );952 953/* Compare a Unicode object with C string and return -1, 0, 1 for less than,954 equal, and greater than, respectively. It is best to pass only955 ASCII-encoded strings, but the function interprets the input string as956 ISO-8859-1 if it contains non-ASCII characters.957 This function does not raise exceptions. */958 959PyAPI_FUNC(int) PyUnicode_CompareWithASCIIString(960 PyObject *left,961 const char *right /* ASCII-encoded string */962 );963 964#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x030D0000965/* Compare a Unicode object with UTF-8 encoded C string.966 Return 1 if they are equal, or 0 otherwise.967 This function does not raise exceptions. */968 969PyAPI_FUNC(int) PyUnicode_EqualToUTF8(PyObject *, const char *);970PyAPI_FUNC(int) PyUnicode_EqualToUTF8AndSize(PyObject *, const char *, Py_ssize_t);971#endif972 973#if !defined(Py_LIMITED_API) || Py_LIMITED_API+0 >= 0x030e0000974PyAPI_FUNC(int) PyUnicode_Equal(PyObject *str1, PyObject *str2);975#endif976 977/* Rich compare two strings and return one of the following:978 979 - NULL in case an exception was raised980 - Py_True or Py_False for successful comparisons981 - Py_NotImplemented in case the type combination is unknown982 983 Possible values for op:984 985 Py_GT, Py_GE, Py_EQ, Py_NE, Py_LT, Py_LE986 987*/988 989PyAPI_FUNC(PyObject *) PyUnicode_RichCompare(990 PyObject *left, /* Left string */991 PyObject *right, /* Right string */992 int op /* Operation: Py_EQ, Py_NE, Py_GT, etc. */993 );994 995/* Apply an argument tuple or dictionary to a format string and return996 the resulting Unicode string. */997 998PyAPI_FUNC(PyObject *) PyUnicode_Format(999 PyObject *format, /* Format string */1000 PyObject *args /* Argument tuple or dictionary */1001 );1002 1003/* Checks whether element is contained in container and return 1/01004 accordingly.1005 1006 element has to coerce to a one element Unicode string. -1 is1007 returned in case of an error. */1008 1009PyAPI_FUNC(int) PyUnicode_Contains(1010 PyObject *container, /* Container string */1011 PyObject *element /* Element string */1012 );1013 1014/* Checks whether argument is a valid identifier. */1015 1016PyAPI_FUNC(int) PyUnicode_IsIdentifier(PyObject *s);1017 1018/* === Characters Type APIs =============================================== */1019 1020#ifndef Py_LIMITED_API1021# define Py_CPYTHON_UNICODEOBJECT_H1022# include "cpython/unicodeobject.h"1023# undef Py_CPYTHON_UNICODEOBJECT_H1024#endif1025 1026#ifdef __cplusplus1027}1028#endif1029#endif /* !Py_UNICODEOBJECT_H */1030 