Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
_doc_common.py1492 linesDownload Raw Back to pypdf
1# Copyright (c) 2006, Mathieu Fenniak2# Copyright (c) 2007, Ashish Kulkarni <kulkarni.ashish@gmail.com>3# Copyright (c) 2024, Pubpub-ZZ4#5# All rights reserved.6#7# Redistribution and use in source and binary forms, with or without8# modification, are permitted provided that the following conditions are9# met:10#11# * Redistributions of source code must retain the above copyright notice,12# this list of conditions and the following disclaimer.13# * Redistributions in binary form must reproduce the above copyright notice,14# this list of conditions and the following disclaimer in the documentation15# and/or other materials provided with the distribution.16# * The name of the author may not be used to endorse or promote products17# derived from this software without specific prior written permission.18#19# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"20# AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE21# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE22# ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE23# LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR24# CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF25# SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS26# INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN27# CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)28# ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE29# POSSIBILITY OF SUCH DAMAGE.30 31import struct32from abc import abstractmethod33from collections.abc import Generator, Iterable, Iterator, Mapping34from datetime import datetime35from typing import (36    Any,37    Optional,38    Union,39    cast,40)41 42from ._encryption import Encryption43from ._page import PageObject, _VirtualList44from ._page_labels import index2label as page_index2page_label45from ._utils import (46    deprecation_with_replacement,47    logger_warning,48    parse_iso8824_date,49)50from .constants import CatalogAttributes as CA51from .constants import CatalogDictionary as CD52from .constants import (53    CheckboxRadioButtonAttributes,54    GoToActionArguments,55    PagesAttributes,56    UserAccessPermissions,57)58from .constants import Core as CO59from .constants import DocumentInformationAttributes as DI60from .constants import FieldDictionaryAttributes as FA61from .constants import PageAttributes as PG62from .errors import PdfReadError, PyPdfError63from .filters import _decompress_with_limit64from .generic import (65    ArrayObject,66    BooleanObject,67    ByteStringObject,68    Destination,69    DictionaryObject,70    EncodedStreamObject,71    Field,72    Fit,73    FloatObject,74    IndirectObject,75    NameObject,76    NullObject,77    NumberObject,78    PdfObject,79    TextStringObject,80    TreeObject,81    ViewerPreferences,82    create_string_object,83    is_null_or_none,84)85from .generic._files import EmbeddedFile86from .types import OutlineType, PagemodeType87from .xmp import XmpInformation88 89 90def convert_to_int(d: bytes, size: int) -> Union[int, tuple[Any, ...]]:91    if size > 8:92        raise PdfReadError("Invalid size in convert_to_int")93    d = b"\x00\x00\x00\x00\x00\x00\x00\x00" + d94    d = d[-8:]95    return cast(int, struct.unpack(">q", d)[0])96 97 98class DocumentInformation(DictionaryObject):99    """100    A class representing the basic document metadata provided in a PDF File.101    This class is accessible through102    :py:class:`PdfReader.metadata<pypdf.PdfReader.metadata>`.103 104    All text properties of the document metadata have105    *two* properties, e.g. author and author_raw. The non-raw property will106    always return a ``TextStringObject``, making it ideal for a case where the107    metadata is being displayed. The raw property can sometimes return a108    ``ByteStringObject``, if pypdf was unable to decode the string's text109    encoding; this requires additional safety in the caller and therefore is not110    as commonly accessed.111    """112 113    def __init__(self) -> None:114        DictionaryObject.__init__(self)115 116    def _get_text(self, key: str) -> Optional[str]:117        retval = self.get(key, None)118        if isinstance(retval, TextStringObject):119            return retval120        if isinstance(retval, ByteStringObject):121            return str(retval)122        return None123 124    @property125    def title(self) -> Optional[str]:126        """127        Read-only property accessing the document's title.128 129        Returns a ``TextStringObject`` or ``None`` if the title is not130        specified.131        """132        return (133            self._get_text(DI.TITLE) or self.get(DI.TITLE).get_object()  # type: ignore134            if self.get(DI.TITLE)135            else None136        )137 138    @property139    def title_raw(self) -> Optional[str]:140        """The "raw" version of title; can return a ``ByteStringObject``."""141        return self.get(DI.TITLE)142 143    @property144    def author(self) -> Optional[str]:145        """146        Read-only property accessing the document's author.147 148        Returns a ``TextStringObject`` or ``None`` if the author is not149        specified.150        """151        return self._get_text(DI.AUTHOR)152 153    @property154    def author_raw(self) -> Optional[str]:155        """The "raw" version of author; can return a ``ByteStringObject``."""156        return self.get(DI.AUTHOR)157 158    @property159    def subject(self) -> Optional[str]:160        """161        Read-only property accessing the document's subject.162 163        Returns a ``TextStringObject`` or ``None`` if the subject is not164        specified.165        """166        return self._get_text(DI.SUBJECT)167 168    @property169    def subject_raw(self) -> Optional[str]:170        """The "raw" version of subject; can return a ``ByteStringObject``."""171        return self.get(DI.SUBJECT)172 173    @property174    def creator(self) -> Optional[str]:175        """176        Read-only property accessing the document's creator.177 178        If the document was converted to PDF from another format, this is the179        name of the application (e.g. OpenOffice) that created the original180        document from which it was converted. Returns a ``TextStringObject`` or181        ``None`` if the creator is not specified.182        """183        return self._get_text(DI.CREATOR)184 185    @property186    def creator_raw(self) -> Optional[str]:187        """The "raw" version of creator; can return a ``ByteStringObject``."""188        return self.get(DI.CREATOR)189 190    @property191    def producer(self) -> Optional[str]:192        """193        Read-only property accessing the document's producer.194 195        If the document was converted to PDF from another format, this is the196        name of the application (for example, macOS Quartz) that converted it to197        PDF. Returns a ``TextStringObject`` or ``None`` if the producer is not198        specified.199        """200        return self._get_text(DI.PRODUCER)201 202    @property203    def producer_raw(self) -> Optional[str]:204        """The "raw" version of producer; can return a ``ByteStringObject``."""205        return self.get(DI.PRODUCER)206 207    @property208    def creation_date(self) -> Optional[datetime]:209        """Read-only property accessing the document's creation date."""210        return parse_iso8824_date(self._get_text(DI.CREATION_DATE))211 212    @property213    def creation_date_raw(self) -> Optional[str]:214        """215        The "raw" version of creation date; can return a ``ByteStringObject``.216 217        Typically in the format ``D:YYYYMMDDhhmmss[+Z-]hh'mm`` where the suffix218        is the offset from UTC.219        """220        return self.get(DI.CREATION_DATE)221 222    @property223    def modification_date(self) -> Optional[datetime]:224        """225        Read-only property accessing the document's modification date.226 227        The date and time the document was most recently modified.228        """229        return parse_iso8824_date(self._get_text(DI.MOD_DATE))230 231    @property232    def modification_date_raw(self) -> Optional[str]:233        """234        The "raw" version of modification date; can return a235        ``ByteStringObject``.236 237        Typically in the format ``D:YYYYMMDDhhmmss[+Z-]hh'mm`` where the suffix238        is the offset from UTC.239        """240        return self.get(DI.MOD_DATE)241 242    @property243    def keywords(self) -> Optional[str]:244        """245        Read-only property accessing the document's keywords.246 247        Returns a ``TextStringObject`` or ``None`` if keywords are not248        specified.249        """250        return self._get_text(DI.KEYWORDS)251 252    @property253    def keywords_raw(self) -> Optional[str]:254        """The "raw" version of keywords; can return a ``ByteStringObject``."""255        return self.get(DI.KEYWORDS)256 257 258class PdfDocCommon:259    """260    Common functions from PdfWriter and PdfReader objects.261 262    This root class is strongly abstracted.263    """264 265    strict: bool = False  # default266 267    flattened_pages: Optional[list[PageObject]] = None268 269    _encryption: Optional[Encryption] = None270 271    _readonly: bool = False272 273    @property274    @abstractmethod275    def root_object(self) -> DictionaryObject:276        ...  # pragma: no cover277 278    @property279    @abstractmethod280    def pdf_header(self) -> str:281        ...  # pragma: no cover282 283    @abstractmethod284    def get_object(285        self, indirect_reference: Union[int, IndirectObject]286    ) -> Optional[PdfObject]:287        ...  # pragma: no cover288 289    @abstractmethod290    def _replace_object(self, indirect: IndirectObject, obj: PdfObject) -> PdfObject:291        ...  # pragma: no cover292 293    @property294    @abstractmethod295    def _info(self) -> Optional[DictionaryObject]:296        ...  # pragma: no cover297 298    @property299    def metadata(self) -> Optional[DocumentInformation]:300        """301        Retrieve the PDF file's document information dictionary, if it exists.302 303        Note that some PDF files use metadata streams instead of document304        information dictionaries, and these metadata streams will not be305        accessed by this function.306        """307        retval = DocumentInformation()308        if self._info is None:309            return None310        retval.update(self._info)311        return retval312 313    @property314    def xmp_metadata(self) -> Optional[XmpInformation]:315        ...  # pragma: no cover316 317    @property318    def viewer_preferences(self) -> Optional[ViewerPreferences]:319        """Returns the existing ViewerPreferences as an overloaded dictionary."""320        o = self.root_object.get(CD.VIEWER_PREFERENCES, None)321        if o is None:322            return None323        o = o.get_object()324        if not isinstance(o, ViewerPreferences):325            o = ViewerPreferences(o)326            if hasattr(o, "indirect_reference") and o.indirect_reference is not None:327                self._replace_object(o.indirect_reference, o)328            else:329                self.root_object[NameObject(CD.VIEWER_PREFERENCES)] = o330        return o331 332    def get_num_pages(self) -> int:333        """334        Calculate the number of pages in this PDF file.335 336        Returns:337            The number of pages of the parsed PDF file.338 339        Raises:340            PdfReadError: If restrictions prevent this action.341 342        """343        # Flattened pages will not work on an encrypted PDF;344        # the PDF file's page count is used in this case. Otherwise,345        # the original method (flattened page count) is used.346        if self.is_encrypted:347            return self.root_object["/Pages"]["/Count"]  # type: ignore348        if self.flattened_pages is None:349            self._flatten(self._readonly)350        assert self.flattened_pages is not None351        return len(self.flattened_pages)352 353    def get_page(self, page_number: int) -> PageObject:354        """355        Retrieve a page by number from this PDF file.356        Most of the time ``.pages[page_number]`` is preferred.357 358        Args:359            page_number: The page number to retrieve360                (pages begin at zero)361 362        Returns:363            A :class:`PageObject<pypdf._page.PageObject>` instance.364 365        """366        if self.flattened_pages is None:367            self._flatten(self._readonly)368        assert self.flattened_pages is not None, "hint for mypy"369        return self.flattened_pages[page_number]370 371    def _get_page_in_node(372        self,373        page_number: int,374    ) -> tuple[DictionaryObject, int]:375        """376        Retrieve the node and position within the /Kids containing the page.377        If page_number is greater than the number of pages, it returns the top node, -1.378        """379        top = cast(DictionaryObject, self.root_object["/Pages"])380 381        def recursive_call(382            node: DictionaryObject, mi: int383        ) -> tuple[Optional[PdfObject], int]:384            ma = cast(int, node.get("/Count", 1))  # default 1 for /Page types385            if node["/Type"] == "/Page":  # type: ignore[comparison-overlap]386                if page_number == mi:387                    return node, -1388                return None, mi + 1389            if (page_number - mi) >= ma:  # not in nodes below390                if node == top:391                    return top, -1392                return None, mi + ma393            for idx, kid in enumerate(cast(ArrayObject, node["/Kids"])):394                kid = cast(DictionaryObject, kid.get_object())395                n, i = recursive_call(kid, mi)396                if n is not None:  # page has just been found ...397                    if i < 0:  # ... just below!398                        return node, idx399                    # ... at lower levels400                    return n, i401                mi = i402            raise PyPdfError("Unexpectedly cannot find the node.")403 404        node, idx = recursive_call(top, 0)405        assert isinstance(node, DictionaryObject), "mypy"406        return node, idx407 408    @property409    def named_destinations(self) -> dict[str, Destination]:410        """A read-only dictionary which maps names to destinations."""411        return self._get_named_destinations()412 413    def get_named_dest_root(self) -> ArrayObject:414        named_dest = ArrayObject()415        if CA.NAMES in self.root_object and isinstance(416            self.root_object[CA.NAMES], DictionaryObject417        ):418            names = cast(DictionaryObject, self.root_object[CA.NAMES])419            if CA.DESTS in names and isinstance(names[CA.DESTS], DictionaryObject):420                # §3.6.3 Name Dictionary (PDF spec 1.7)421                dests = cast(DictionaryObject, names[CA.DESTS])422                dests_ref = dests.indirect_reference423                if CA.NAMES in dests:424                    # §7.9.6, entries in a name tree node dictionary425                    named_dest = cast(ArrayObject, dests[CA.NAMES])426                else:427                    named_dest = ArrayObject()428                    dests[NameObject(CA.NAMES)] = named_dest429            elif hasattr(self, "_add_object"):430                dests = DictionaryObject()431                dests_ref = self._add_object(dests)432                names[NameObject(CA.DESTS)] = dests_ref433                dests[NameObject(CA.NAMES)] = named_dest434 435        elif hasattr(self, "_add_object"):436            names = DictionaryObject()437            names_ref = self._add_object(names)438            self.root_object[NameObject(CA.NAMES)] = names_ref439            dests = DictionaryObject()440            dests_ref = self._add_object(dests)441            names[NameObject(CA.DESTS)] = dests_ref442            dests[NameObject(CA.NAMES)] = named_dest443 444        return named_dest445 446    ## common447    def _get_named_destinations(448        self,449        tree: Union[TreeObject, None] = None,450        retval: Optional[dict[str, Destination]] = None,451    ) -> dict[str, Destination]:452        """453        Retrieve the named destinations present in the document.454 455        Args:456            tree: The current tree.457            retval: The previously retrieved destinations for nested calls.458 459        Returns:460            A dictionary which maps names to destinations.461 462        """463        if retval is None:464            retval = {}465            catalog = self.root_object466 467            # get the name tree468            if CA.DESTS in catalog:469                tree = cast(TreeObject, catalog[CA.DESTS])470            elif CA.NAMES in catalog:471                names = cast(DictionaryObject, catalog[CA.NAMES])472                if CA.DESTS in names:473                    tree = cast(TreeObject, names[CA.DESTS])474 475        if is_null_or_none(tree):476            return retval477        assert tree is not None, "mypy"478 479        if PagesAttributes.KIDS in tree:480            # recurse down the tree481            for kid in cast(ArrayObject, tree[PagesAttributes.KIDS]):482                self._get_named_destinations(kid.get_object(), retval)483        # §7.9.6, entries in a name tree node dictionary484        elif CA.NAMES in tree:  # /Kids and /Names are exclusives (§7.9.6)485            names = cast(DictionaryObject, tree[CA.NAMES])486            i = 0487            while i < len(names):488                key = names[i].get_object()489                i += 1490                if not isinstance(key, (bytes, str)):491                    continue492                try:493                    value = names[i].get_object()494                except IndexError:495                    break496                i += 1497                if isinstance(value, DictionaryObject):498                    if "/D" in value:499                        value = value["/D"]500                    else:501                        continue502                dest = self._build_destination(key, value)503                if dest is not None:504                    retval[cast(str, dest["/Title"])] = dest505                    # Remain backwards-compatible.506                    retval[str(key)] = dest507        else:  # case where Dests is in root catalog (PDF 1.7 specs, §2 about PDF 1.1)508            for k__, v__ in tree.items():509                val = v__.get_object()510                if isinstance(val, DictionaryObject):511                    if "/D" in val:512                        val = val["/D"].get_object()513                    else:514                        continue515                dest = self._build_destination(k__, val)516                if dest is not None:517                    retval[k__] = dest518        return retval519 520    # A select group of relevant field attributes. For the complete list,521    # see §12.3.2 of the PDF 1.7 or PDF 2.0 specification.522 523    def get_fields(524        self,525        tree: Optional[TreeObject] = None,526        retval: Optional[dict[Any, Any]] = None,527        fileobj: Optional[Any] = None,528        stack: Optional[list[PdfObject]] = None,529    ) -> Optional[dict[str, Any]]:530        """531        Extract field data if this PDF contains interactive form fields.532 533        The *tree*, *retval*, *stack* parameters are for recursive use.534 535        Args:536            tree: Current object to parse.537            retval: In-progress list of fields.538            fileobj: A file object (usually a text file) to write539                a report to on all interactive form fields found.540            stack: List of already parsed objects.541 542        Returns:543            A dictionary where each key is a field name, and each544            value is a :class:`Field<pypdf.generic.Field>` object. By545            default, the mapping name is used for keys.546            ``None`` if form data could not be located.547 548        """549        field_attributes = FA.attributes_dict()550        field_attributes.update(CheckboxRadioButtonAttributes.attributes_dict())551        if retval is None:552            retval = {}553            catalog = self.root_object554            stack = []555            # get the AcroForm tree556            if CD.ACRO_FORM in catalog:557                tree = cast(Optional[TreeObject], catalog[CD.ACRO_FORM])558            else:559                return None560        if tree is None:561            return retval562        assert stack is not None563        if "/Fields" in tree:564            fields = cast(ArrayObject, tree["/Fields"])565            for f in fields:566                field = f.get_object()567                self._build_field(field, retval, fileobj, field_attributes, stack)568        elif any(attr in tree for attr in field_attributes):569            # Tree is a field570            self._build_field(tree, retval, fileobj, field_attributes, stack)571        return retval572 573    def _get_qualified_field_name(self, parent: DictionaryObject) -> str:574        if "/TM" in parent:575            return cast(str, parent["/TM"])576        if "/Parent" in parent:577            return (578                self._get_qualified_field_name(579                    cast(DictionaryObject, parent["/Parent"])580                )581                + "."582                + cast(str, parent.get("/T", ""))583            )584        return cast(str, parent.get("/T", ""))585 586    def _build_field(587        self,588        field: Union[TreeObject, DictionaryObject],589        retval: dict[Any, Any],590        fileobj: Any,591        field_attributes: Any,592        stack: list[PdfObject],593    ) -> None:594        if all(attr not in field for attr in ("/T", "/TM")):595            return596        key = self._get_qualified_field_name(field)597        if fileobj:598            self._write_field(fileobj, field, field_attributes)599            fileobj.write("\n")600        retval[key] = Field(field)601        obj = retval[key].indirect_reference.get_object()  # to get the full object602        if obj.get(FA.FT, "") == "/Ch" and obj.get(NameObject(FA.Opt)):603            retval[key][NameObject("/_States_")] = obj[NameObject(FA.Opt)]604        if obj.get(FA.FT, "") == "/Btn" and "/AP" in obj:605            #  Checkbox606            retval[key][NameObject("/_States_")] = ArrayObject(607                list(obj["/AP"]["/N"].keys())608            )609            if "/Off" not in retval[key]["/_States_"]:610                retval[key][NameObject("/_States_")].append(NameObject("/Off"))611        elif obj.get(FA.FT, "") == "/Btn" and obj.get(FA.Ff, 0) & FA.FfBits.Radio != 0:612            states: list[str] = []613            retval[key][NameObject("/_States_")] = ArrayObject(states)614            for k in obj.get(FA.Kids, {}):615                k = k.get_object()616                for s in list(k["/AP"]["/N"].keys()):617                    if s not in states:618                        states.append(s)619                retval[key][NameObject("/_States_")] = ArrayObject(states)620            if (621                obj.get(FA.Ff, 0) & FA.FfBits.NoToggleToOff != 0622                and "/Off" in retval[key]["/_States_"]623            ):624                del retval[key]["/_States_"][retval[key]["/_States_"].index("/Off")]625        # at last for order626        self._check_kids(field, retval, fileobj, stack)627 628    def _check_kids(629        self,630        tree: Union[TreeObject, DictionaryObject],631        retval: Any,632        fileobj: Any,633        stack: list[PdfObject],634    ) -> None:635        if tree in stack:636            logger_warning(637                f"{self._get_qualified_field_name(tree)} already parsed", __name__638            )639            return640        stack.append(tree)641        if PagesAttributes.KIDS in tree:642            # recurse down the tree643            for kid in tree[PagesAttributes.KIDS]:  # type: ignore644                kid = kid.get_object()645                self.get_fields(kid, retval, fileobj, stack)646 647    def _write_field(self, fileobj: Any, field: Any, field_attributes: Any) -> None:648        field_attributes_tuple = FA.attributes()649        field_attributes_tuple = (650            field_attributes_tuple + CheckboxRadioButtonAttributes.attributes()651        )652 653        for attr in field_attributes_tuple:654            if attr in (655                FA.Kids,656                FA.AA,657            ):658                continue659            attr_name = field_attributes[attr]660            try:661                if attr == FA.FT:662                    # Make the field type value clearer663                    types = {664                        "/Btn": "Button",665                        "/Tx": "Text",666                        "/Ch": "Choice",667                        "/Sig": "Signature",668                    }669                    if field[attr] in types:670                        fileobj.write(f"{attr_name}: {types[field[attr]]}\n")671                elif attr == FA.Parent:672                    # Let's just write the name of the parent673                    try:674                        name = field[attr][FA.TM]675                    except KeyError:676                        name = field[attr][FA.T]677                    fileobj.write(f"{attr_name}: {name}\n")678                else:679                    fileobj.write(f"{attr_name}: {field[attr]}\n")680            except KeyError:681                # Field attribute is N/A or unknown, so don't write anything682                pass683 684    def get_form_text_fields(self, full_qualified_name: bool = False) -> dict[str, Any]:685        """686        Retrieve form fields from the document with textual data.687 688        Args:689            full_qualified_name: to get full name690 691        Returns:692            A dictionary. The key is the name of the form field,693            the value is the content of the field.694 695            If the document contains multiple form fields with the same name, the696            second and following will get the suffix .2, .3, ...697 698        """699 700        def indexed_key(k: str, fields: dict[Any, Any]) -> str:701            if k not in fields:702                return k703            return (704                k705                + "."706                + str(sum(1 for kk in fields if kk.startswith(k + ".")) + 2)707            )708 709        # Retrieve document form fields710        formfields = self.get_fields()711        if formfields is None:712            return {}713        ff = {}714        for field, value in formfields.items():715            if value.get("/FT") == "/Tx":716                if full_qualified_name:717                    ff[field] = value.get("/V")718                else:719                    ff[indexed_key(cast(str, value["/T"]), ff)] = value.get("/V")720        return ff721 722    def get_pages_showing_field(723        self, field: Union[Field, PdfObject, IndirectObject]724    ) -> list[PageObject]:725        """726        Provides list of pages where the field is called.727 728        Args:729            field: Field Object, PdfObject or IndirectObject referencing a Field730 731        Returns:732            List of pages:733                - Empty list:734                    The field has no widgets attached735                    (either hidden field or ancestor field).736                - Single page list:737                    Page where the widget is present738                    (most common).739                - Multi-page list:740                    Field with multiple kids widgets741                    (example: radio buttons, field repeated on multiple pages).742 743        """744 745        def _get_inherited(obj: DictionaryObject, key: str) -> Any:746            if key in obj:747                return obj[key]748            if "/Parent" in obj:749                return _get_inherited(750                    cast(DictionaryObject, obj["/Parent"].get_object()), key751                )752            return None753 754        try:755            # to cope with all types756            field = cast(DictionaryObject, field.indirect_reference.get_object())  # type: ignore757        except Exception as exc:758            raise ValueError("Field type is invalid") from exc759        if is_null_or_none(_get_inherited(field, "/FT")):760            raise ValueError("Field is not valid")761        ret = []762        if field.get("/Subtype", "") == "/Widget":763            if "/P" in field:764                ret = [field["/P"].get_object()]765            else:766                ret = [767                    p768                    for p in self.pages769                    if field.indirect_reference in p.get("/Annots", "")770                ]771        else:772            kids = field.get("/Kids", ())773            for k in kids:774                k = k.get_object()775                if (k.get("/Subtype", "") == "/Widget") and ("/T" not in k):776                    # Kid that is just a widget, not a field:777                    if "/P" in k:778                        ret += [k["/P"].get_object()]779                    else:780                        ret += [781                            p782                            for p in self.pages783                            if k.indirect_reference in p.get("/Annots", "")784                        ]785        return [786            x787            if isinstance(x, PageObject)788            else (self.pages[self._get_page_number_by_indirect(x.indirect_reference)])  # type: ignore789            for x in ret790        ]791 792    @property793    def open_destination(794        self,795    ) -> Union[None, Destination, TextStringObject, ByteStringObject]:796        """797        Property to access the opening destination (``/OpenAction`` entry in798        the PDF catalog). It returns ``None`` if the entry does not exist799        or is not set.800 801        Raises:802            Exception: If a destination is invalid.803 804        """805        if "/OpenAction" not in self.root_object:806            return None807        oa: Any = self.root_object["/OpenAction"]808        if isinstance(oa, bytes):  # pragma: no cover809            oa = oa.decode()810        if isinstance(oa, str):811            return create_string_object(oa)812        if isinstance(oa, ArrayObject):813            try:814                page, typ, *array = oa815                fit = Fit(typ, tuple(array))816                return Destination("OpenAction", page, fit)817            except Exception as exc:818                raise Exception(f"Invalid Destination {oa}: {exc}")819        else:820            return None821 822    @open_destination.setter823    def open_destination(self, dest: Union[None, str, Destination, PageObject]) -> None:824        raise NotImplementedError("No setter for open_destination")825 826    @property827    def outline(self) -> OutlineType:828        """829        Read-only property for the outline present in the document830        (i.e., a collection of 'outline items' which are also known as831        'bookmarks').832        """833        return self._get_outline()834 835    def _get_outline(836        self,837        node: Optional[DictionaryObject] = None,838        outline: Optional[Any] = None,839        visited: Optional[set[int]] = None,840    ) -> OutlineType:841        if outline is None:842            outline = []843            catalog = self.root_object844 845            # get the outline dictionary and named destinations846            if CO.OUTLINES in catalog:847                lines = cast(DictionaryObject, catalog[CO.OUTLINES])848 849                if isinstance(lines, NullObject):850                    return outline851 852                # §12.3.3 Document outline, entries in the outline dictionary853                if not is_null_or_none(lines) and "/First" in lines:854                    node = cast(DictionaryObject, lines["/First"])855            self._named_destinations = self._get_named_destinations()856 857        if node is None:858            return outline859 860        # see if there are any more outline items861        if visited is None:862            visited = set()863        while True:864            node_id = id(node)865            if node_id in visited:866                logger_warning(f"Detected cycle in outline structure for {node}", __name__)867                break868            visited.add(node_id)869 870            outline_obj = self._build_outline_item(node)871            if outline_obj:872                outline.append(outline_obj)873 874            # check for sub-outline875            if "/First" in node:876                sub_outline: list[Any] = []877                # Pass a copy to allow multiple outer entries to reference the same inner one.878                inner_visited = visited.copy()879                self._get_outline(880                    node=cast(DictionaryObject, node["/First"]),881                    outline=sub_outline,882                    visited=inner_visited,883                )884                if sub_outline:885                    outline.append(sub_outline)886 887            if "/Next" not in node:888                break889            node = cast(DictionaryObject, node["/Next"])890 891        return outline892 893    @property894    def threads(self) -> Optional[ArrayObject]:895        """896        Read-only property for the list of threads.897 898        See §12.4.3 from the PDF 1.7 or 2.0 specification.899 900        It is an array of dictionaries with "/F" (the first bead in the thread)901        and "/I" (a thread information dictionary containing information about902        the thread, such as its title, author, and creation date) properties or903        None if there are no articles.904 905        Since PDF 2.0 it can also contain an indirect reference to a metadata906        stream containing information about the thread, such as its title,907        author, and creation date.908        """909        catalog = self.root_object910        if CO.THREADS in catalog:911            return cast("ArrayObject", catalog[CO.THREADS])912        return None913 914    @abstractmethod915    def _get_page_number_by_indirect(916        self, indirect_reference: Union[None, int, NullObject, IndirectObject]917    ) -> Optional[int]:918        ...  # pragma: no cover919 920    def get_page_number(self, page: PageObject) -> Optional[int]:921        """922        Retrieve page number of a given PageObject.923 924        Args:925            page: The page to get page number. Should be926                an instance of :class:`PageObject<pypdf._page.PageObject>`927 928        Returns:929            The page number or None if page is not found930 931        """932        return self._get_page_number_by_indirect(page.indirect_reference)933 934    def get_destination_page_number(self, destination: Destination) -> Optional[int]:935        """936        Retrieve page number of a given Destination object.937 938        Args:939            destination: The destination to get page number.940 941        Returns:942            The page number or None if page is not found943 944        """945        return self._get_page_number_by_indirect(destination.page)946 947    def _build_destination(948        self,949        title: Union[str, bytes],950        array: Optional[951            list[952                Union[NumberObject, IndirectObject, None, NullObject, DictionaryObject]953            ]954        ],955    ) -> Destination:956        page, typ = None, None957        # handle outline items with missing or invalid destination958        if (959            isinstance(array, (NullObject, str))960            or (isinstance(array, ArrayObject) and len(array) == 0)961            or array is None962        ):963            page = NullObject()964            return Destination(title, page, Fit.fit())965        page, typ, *array = array  # type: ignore966        try:967            return Destination(title, page, Fit(fit_type=typ, fit_args=array))  # type: ignore968        except PdfReadError:969            logger_warning(f"Unknown destination: {title!r} {array}", __name__)970            if self.strict:971                raise972            # create a link to first Page973            tmp = self.pages[0].indirect_reference974            indirect_reference = NullObject() if tmp is None else tmp975            return Destination(title, indirect_reference, Fit.fit())976 977    def _build_outline_item(self, node: DictionaryObject) -> Optional[Destination]:978        dest, title, outline_item = None, None, None979 980        # title required for valid outline981        # §12.3.3, entries in an outline item dictionary982        try:983            title = cast("str", node["/Title"])984        except KeyError:985            if self.strict:986                raise PdfReadError(f"Outline Entry Missing /Title attribute: {node!r}")987            title = ""988 989        if "/A" in node:990            # Action, PDF 1.7 and PDF 2.0 §12.6 (only type GoTo supported)991            action = cast(DictionaryObject, node["/A"])992            action_type = cast(NameObject, action[GoToActionArguments.S])993            if action_type == "/GoTo":994                if GoToActionArguments.D in action:995                    dest = action[GoToActionArguments.D]996                elif self.strict:997                    raise PdfReadError(f"Outline Action Missing /D attribute: {node!r}")998        elif "/Dest" in node:999            # Destination, PDF 1.7 and PDF 2.0 §12.3.21000            dest = node["/Dest"]1001            # if array was referenced in another object, will be a dict w/ key "/D"1002            if isinstance(dest, DictionaryObject) and "/D" in dest:1003                dest = dest["/D"]1004 1005        if isinstance(dest, ArrayObject):1006            outline_item = self._build_destination(title, dest)1007        elif isinstance(dest, str):1008            # named destination, addresses NameObject Issue #1931009            # TODO: Keep named destination instead of replacing it?1010            try:1011                outline_item = self._build_destination(1012                    title, self._named_destinations[dest].dest_array1013                )1014            except KeyError:1015                # named destination not found in Name Dict1016                outline_item = self._build_destination(title, None)1017        elif dest is None:1018            # outline item not required to have destination or action1019            # PDFv1.7 Table 1531020            outline_item = self._build_destination(title, dest)1021        else:1022            if self.strict:1023                raise PdfReadError(f"Unexpected destination {dest!r}")1024            logger_warning(1025                f"Removed unexpected destination {dest!r} from destination",1026                __name__,1027            )1028            outline_item = self._build_destination(title, None)1029 1030        # if outline item created, add color, format, and child count if present1031        if outline_item:1032            if "/C" in node:1033                # Color of outline item font in (R, G, B) with values ranging 0.0-1.01034                outline_item[NameObject("/C")] = ArrayObject(FloatObject(c) for c in node["/C"])  # type: ignore1035            if "/F" in node:1036                # specifies style characteristics bold and/or italic1037                # with 1=italic, 2=bold, 3=both1038                outline_item[NameObject("/F")] = node["/F"]1039            if "/Count" in node:1040                # absolute value = num. visible children1041                # with positive = open/unfolded, negative = closed/folded1042                outline_item[NameObject("/Count")] = node["/Count"]1043            #  if count is 0 we will consider it as open (to have available is_open)1044            outline_item[NameObject("/%is_open%")] = BooleanObject(1045                node.get("/Count", 0) >= 01046            )1047        outline_item.node = node1048        try:1049            outline_item.indirect_reference = node.indirect_reference1050        except AttributeError:1051            pass1052        return outline_item1053 1054    @property1055    def pages(self) -> list[PageObject]:1056        """1057        Property that emulates a list of :class:`PageObject<pypdf._page.PageObject>`.1058        This property allows to get a page or a range of pages.1059 1060        Note:1061            For PdfWriter only: Provides the capability to remove a page/range of1062            page from the list (using the del operator). Remember: Only the page1063            entry is removed, as the objects beneath can be used elsewhere. A1064            solution to completely remove them - if they are not used anywhere - is1065            to write to a buffer/temporary file and then load it into a new1066            PdfWriter.1067 1068        """1069        return _VirtualList(self.get_num_pages, self.get_page)  # type: ignore1070 1071    @property1072    def page_labels(self) -> list[str]:1073        """1074        A list of labels for the pages in this document.1075 1076        This property is read-only. The labels are in the order that the pages1077        appear in the document.1078        """1079        return [page_index2page_label(self, i) for i in range(len(self.pages))]1080 1081    @property1082    def page_layout(self) -> Optional[str]:1083        """1084        Get the page layout currently being used.1085 1086        .. list-table:: Valid ``layout`` values1087           :widths: 50 2001088 1089           * - /NoLayout1090             - Layout explicitly not specified1091           * - /SinglePage1092             - Show one page at a time1093           * - /OneColumn1094             - Show one column at a time1095           * - /TwoColumnLeft1096             - Show pages in two columns, odd-numbered pages on the left1097           * - /TwoColumnRight1098             - Show pages in two columns, odd-numbered pages on the right1099           * - /TwoPageLeft1100             - Show two pages at a time, odd-numbered pages on the left1101           * - /TwoPageRight1102             - Show two pages at a time, odd-numbered pages on the right1103        """1104        try:1105            return cast(NameObject, self.root_object[CD.PAGE_LAYOUT])1106        except KeyError:1107            return None1108 1109    @property1110    def page_mode(self) -> Optional[PagemodeType]:1111        """1112        Get the page mode currently being used.1113 1114        .. list-table:: Valid ``mode`` values1115           :widths: 50 2001116 1117           * - /UseNone1118             - Do not show outline or thumbnails panels1119           * - /UseOutlines1120             - Show outline (aka bookmarks) panel1121           * - /UseThumbs1122             - Show page thumbnails panel1123           * - /FullScreen1124             - Fullscreen view1125           * - /UseOC1126             - Show Optional Content Group (OCG) panel1127           * - /UseAttachments1128             - Show attachments panel1129        """1130        try:1131            return self.root_object["/PageMode"]  # type: ignore1132        except KeyError:1133            return None1134 1135    def _flatten(1136        self,1137        list_only: bool = False,1138        pages: Union[None, DictionaryObject, PageObject] = None,1139        inherit: Optional[dict[str, Any]] = None,1140        indirect_reference: Optional[IndirectObject] = None,1141    ) -> None:1142        """1143        Process the document pages to ease searching.1144 1145        Attributes of a page may inherit from ancestor nodes1146        in the page tree. Flattening means moving1147        any inheritance data into descendant nodes,1148        effectively removing the inheritance dependency.1149 1150        Note: It is distinct from another use of "flattening" applied to PDFs.1151        Flattening a PDF also means combining all the contents into one single layer1152        and making the file less editable.1153 1154        Args:1155            list_only: Will only list the pages within _flatten_pages.1156            pages:1157            inherit:1158            indirect_reference: Used recursively to flatten the /Pages object.1159 1160        """1161        inheritable_page_attributes = (1162            NameObject(PG.RESOURCES),1163            NameObject(PG.MEDIABOX),1164            NameObject(PG.CROPBOX),1165            NameObject(PG.ROTATE),1166        )1167        if inherit is None:1168            inherit = {}1169        if is_null_or_none(pages):1170            # Fix issue 327: set flattened_pages attribute only for1171            # decrypted file1172            catalog = self.root_object1173            pages = catalog.get("/Pages").get_object()  # type: ignore1174            if not isinstance(pages, DictionaryObject):1175                raise PdfReadError("Invalid object in /Pages")1176            self.flattened_pages = []1177        assert pages is not None, "mypy"1178 1179        if PagesAttributes.TYPE in pages:1180            t = cast(str, pages[PagesAttributes.TYPE])1181        # if the page tree node has no /Type, consider as a page if /Kids is also missing1182        elif PagesAttributes.KIDS not in pages:1183            t = "/Page"1184        else:1185            t = "/Pages"1186 1187        if t == "/Pages":1188            for attr in inheritable_page_attributes:1189                if attr in pages:1190                    inherit[attr] = pages[attr]1191            pages_reference = getattr(pages, "indirect_reference", object())1192            for page in cast(ArrayObject, pages[PagesAttributes.KIDS]):1193                if getattr(page, "indirect_reference", object()) == pages_reference:1194                    raise PdfReadError("Detected cyclic page references.")1195 1196                addt = {}1197                if isinstance(page, IndirectObject):1198                    addt["indirect_reference"] = page1199                obj = page.get_object()1200                if obj:

Showing the first 1,200 of 1492 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai