Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
_page.py2357 linesDownload Raw Back to pypdf
1# Copyright (c) 2006, Mathieu Fenniak2# Copyright (c) 2007, Ashish Kulkarni <kulkarni.ashish@gmail.com>3#4# All rights reserved.5#6# Redistribution and use in source and binary forms, with or without7# modification, are permitted provided that the following conditions are8# met:9#10# * Redistributions of source code must retain the above copyright notice,11# this list of conditions and the following disclaimer.12# * Redistributions in binary form must reproduce the above copyright notice,13# this list of conditions and the following disclaimer in the documentation14# and/or other materials provided with the distribution.15# * The name of the author may not be used to endorse or promote products16# derived from this software without specific prior written permission.17#18# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"19# AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE20# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE21# ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE22# LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR23# CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF24# SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS25# INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN26# CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)27# ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE28# POSSIBILITY OF SUCH DAMAGE.29 30import math31from collections.abc import Iterable, Iterator, Sequence32from copy import deepcopy33from dataclasses import asdict, dataclass34from decimal import Decimal35from io import BytesIO36from pathlib import Path37from typing import (38    Any,39    Callable,40    Literal,41    Optional,42    Union,43    cast,44    overload,45)46 47from ._font import Font48from ._protocols import PdfCommonDocProtocol49from ._text_extraction import (50    _layout_mode,51)52from ._text_extraction._text_extractor import TextExtraction53from ._utils import (54    CompressedTransformationMatrix,55    TransformationMatrixType,56    _human_readable_bytes,57    deprecate,58    logger_warning,59    matrix_multiply,60)61from .constants import (62    _INLINE_IMAGE_KEY_MAPPING,63    _INLINE_IMAGE_VALUE_MAPPING,64    AnnotationDictionaryAttributes,65    ImageAttributes,66)67from .constants import PageAttributes as PG68from .constants import Resources as RES69from .errors import PageSizeNotDefinedError, PdfReadError70from .generic import (71    ArrayObject,72    ContentStream,73    DictionaryObject,74    EncodedStreamObject,75    FloatObject,76    IndirectObject,77    NameObject,78    NullObject,79    NumberObject,80    PdfObject,81    RectangleObject,82    StreamObject,83    is_null_or_none,84)85 86try:87    from PIL.Image import Image88 89    pil_not_imported = False90except ImportError:91    Image = object  # type: ignore[assignment,misc,unused-ignore]  # TODO: Remove unused-ignore on Python 3.1092    pil_not_imported = True  # error will be raised only when using images93 94MERGE_CROP_BOX = "cropbox"  # pypdf <= 3.4.0 used "trimbox"95 96 97def _get_rectangle(self: Any, name: str, defaults: Iterable[str]) -> RectangleObject:98    retval: Union[None, RectangleObject, ArrayObject, IndirectObject] = self.get(name)99    if isinstance(retval, RectangleObject):100        return retval101    if is_null_or_none(retval):102        for d in defaults:103            retval = self.get(d)104            if retval is not None:105                break106    if isinstance(retval, IndirectObject):107        retval = self.pdf.get_object(retval)108    if isinstance(retval, ArrayObject) and (length := len(retval)) > 4:109        logger_warning(f"Expected four values, got {length}: {retval}", __name__)110        retval = RectangleObject(tuple(retval[:4]))111    else:112        retval = RectangleObject(retval)  # type: ignore113    _set_rectangle(self, name, retval)114    return retval115 116 117def _set_rectangle(self: Any, name: str, value: Union[RectangleObject, float]) -> None:118    self[NameObject(name)] = value119 120 121def _delete_rectangle(self: Any, name: str) -> None:122    del self[name]123 124 125def _create_rectangle_accessor(name: str, fallback: Iterable[str]) -> property:126    return property(127        lambda self: _get_rectangle(self, name, fallback),128        lambda self, value: _set_rectangle(self, name, value),129        lambda self: _delete_rectangle(self, name),130    )131 132 133class Transformation:134    """135    Represent a 2D transformation.136 137    The transformation between two coordinate systems is represented by a 3-by-3138    transformation matrix with the following form::139 140        a b 0141        c d 0142        e f 1143 144    Because a transformation matrix has only six elements that can be changed,145    it is usually specified in PDF as the six-element array [ a b c d e f ].146 147    Coordinate transformations are expressed as matrix multiplications::148 149                                 a b 0150     [ x′ y′ 1 ] = [ x y 1 ] ×   c d 0151                                 e f 1152 153 154    Example:155        >>> from pypdf import PdfWriter, Transformation156        >>> page = PdfWriter().add_blank_page(800, 600)157        >>> op = Transformation().scale(sx=2, sy=3).translate(tx=10, ty=20)158        >>> page.add_transformation(op)159 160    """161 162    def __init__(self, ctm: CompressedTransformationMatrix = (1, 0, 0, 1, 0, 0)) -> None:163        self.ctm = ctm164 165    @property166    def matrix(self) -> TransformationMatrixType:167        """168        Return the transformation matrix as a tuple of tuples in the form:169 170        ((a, b, 0), (c, d, 0), (e, f, 1))171        """172        return (173            (self.ctm[0], self.ctm[1], 0),174            (self.ctm[2], self.ctm[3], 0),175            (self.ctm[4], self.ctm[5], 1),176        )177 178    @staticmethod179    def compress(matrix: TransformationMatrixType) -> CompressedTransformationMatrix:180        """181        Compresses the transformation matrix into a tuple of (a, b, c, d, e, f).182 183        Args:184            matrix: The transformation matrix as a tuple of tuples.185 186        Returns:187            A tuple representing the transformation matrix as (a, b, c, d, e, f)188 189        """190        return (191            matrix[0][0],192            matrix[0][1],193            matrix[1][0],194            matrix[1][1],195            matrix[2][0],196            matrix[2][1],197        )198 199    def _to_cm(self) -> str:200        # Returns the cm operation string for the given transformation matrix201        return (202            f"{self.ctm[0]:.4f} {self.ctm[1]:.4f} {self.ctm[2]:.4f} "203            f"{self.ctm[3]:.4f} {self.ctm[4]:.4f} {self.ctm[5]:.4f} cm"204        )205 206    def transform(self, m: "Transformation") -> "Transformation":207        """208        Apply one transformation to another.209 210        Args:211            m: a Transformation to apply.212 213        Returns:214            A new ``Transformation`` instance215 216        Example:217            >>> from pypdf import PdfWriter, Transformation218            >>> height, width = 40, 50219            >>> page = PdfWriter().add_blank_page(800, 600)220            >>> op = Transformation((1, 0, 0, -1, 0, height)) # vertical mirror221            >>> op = Transformation().transform(Transformation((-1, 0, 0, 1, width, 0)))  # horizontal mirror222            >>> page.add_transformation(op)223 224        """225        ctm = Transformation.compress(matrix_multiply(self.matrix, m.matrix))226        return Transformation(ctm)227 228    def translate(self, tx: float = 0, ty: float = 0) -> "Transformation":229        """230        Translate the contents of a page.231 232        Args:233            tx: The translation along the x-axis.234            ty: The translation along the y-axis.235 236        Returns:237            A new ``Transformation`` instance238 239        """240        m = self.ctm241        return Transformation(ctm=(m[0], m[1], m[2], m[3], m[4] + tx, m[5] + ty))242 243    def scale(244        self, sx: Optional[float] = None, sy: Optional[float] = None245    ) -> "Transformation":246        """247        Scale the contents of a page towards the origin of the coordinate system.248 249        Typically, that is the lower-left corner of the page. That can be250        changed by translating the contents / the page boxes.251 252        Args:253            sx: The scale factor along the x-axis.254            sy: The scale factor along the y-axis.255 256        Returns:257            A new Transformation instance with the scaled matrix.258 259        """260        if sx is None and sy is None:261            raise ValueError("Either sx or sy must be specified")262        if sx is None:263            sx = sy264        if sy is None:265            sy = sx266        assert sx is not None267        assert sy is not None268        op: TransformationMatrixType = ((sx, 0, 0), (0, sy, 0), (0, 0, 1))269        ctm = Transformation.compress(matrix_multiply(self.matrix, op))270        return Transformation(ctm)271 272    def rotate(self, rotation: float) -> "Transformation":273        """274        Rotate the contents of a page.275 276        Args:277            rotation: The angle of rotation in degrees.278 279        Returns:280            A new ``Transformation`` instance with the rotated matrix.281 282        """283        rotation = math.radians(rotation)284        op: TransformationMatrixType = (285            (math.cos(rotation), math.sin(rotation), 0),286            (-math.sin(rotation), math.cos(rotation), 0),287            (0, 0, 1),288        )289        ctm = Transformation.compress(matrix_multiply(self.matrix, op))290        return Transformation(ctm)291 292    def __repr__(self) -> str:293        return f"Transformation(ctm={self.ctm})"294 295    @overload296    def apply_on(self, pt: list[float], as_object: bool = False) -> list[float]:297        ...298 299    @overload300    def apply_on(301        self, pt: tuple[float, float], as_object: bool = False302    ) -> tuple[float, float]:303        ...304 305    def apply_on(306        self,307        pt: Union[tuple[float, float], list[float]],308        as_object: bool = False,309    ) -> Union[tuple[float, float], list[float]]:310        """311        Apply the transformation matrix on the given point.312 313        Args:314            pt: A tuple or list representing the point in the form (x, y).315            as_object: If True, return items as FloatObject, otherwise as plain floats.316 317        Returns:318            A tuple or list representing the transformed point in the form (x', y')319 320        """321        typ = FloatObject if as_object else float322        pt1 = (323            typ(float(pt[0]) * self.ctm[0] + float(pt[1]) * self.ctm[2] + self.ctm[4]),324            typ(float(pt[0]) * self.ctm[1] + float(pt[1]) * self.ctm[3] + self.ctm[5]),325        )326        return list(pt1) if isinstance(pt, list) else pt1327 328 329@dataclass330class ImageFile:331    """332    Image within the PDF file. *This object is not designed to be built.*333 334    This object should not be modified except using :func:`ImageFile.replace` to replace the image with a new one.335    """336 337    name: str = ""338    """339    Filename as identified within the PDF file.340    """341 342    data: bytes = b""343    """344    Data as bytes.345    """346 347    image: Optional[Image] = None348    """349    Data as PIL image.350    """351 352    indirect_reference: Optional[IndirectObject] = None353    """354    Reference to the object storing the stream.355    """356 357    def replace(self, new_image: Image, **kwargs: Any) -> None:358        """359        Replace the image with a new PIL image.360 361        Args:362            new_image (PIL.Image.Image): The new PIL image to replace the existing image.363            **kwargs: Additional keyword arguments to pass to `Image.save()`.364 365        Raises:366            TypeError: If the image is inline or in a PdfReader.367            TypeError: If the image does not belong to a PdfWriter.368            TypeError: If `new_image` is not a PIL Image.369 370        Note:371            This method replaces the existing image with a new image.372            It is not allowed for inline images or images within a PdfReader.373            The `kwargs` parameter allows passing additional parameters374            to `Image.save()`, such as quality.375 376        """377        if pil_not_imported:378            raise ImportError(379                "pillow is required to do image extraction. "380                "It can be installed via 'pip install pypdf[image]'"381            )382 383        from ._reader import PdfReader  # noqa: PLC0415384        from .generic import DictionaryObject, PdfObject  # noqa: PLC0415385        from .generic._image_xobject import _xobj_to_image  # noqa: PLC0415386 387        if self.indirect_reference is None:388            raise TypeError("Cannot update an inline image.")389        if not hasattr(self.indirect_reference.pdf, "_id_translated"):390            raise TypeError("Cannot update an image not belonging to a PdfWriter.")391        if not isinstance(new_image, Image):392            raise TypeError("new_image shall be a PIL Image")393        b = BytesIO()394        new_image.save(b, "PDF", **kwargs)395        reader = PdfReader(b)396        page_image = reader.pages[0].images[0]397        assert page_image.indirect_reference is not None398        self.indirect_reference.pdf._objects[self.indirect_reference.idnum - 1] = (399            page_image.indirect_reference.get_object()400        )401        cast(402            PdfObject, self.indirect_reference.get_object()403        ).indirect_reference = self.indirect_reference404        # change the object attributes405        extension, byte_stream, img = _xobj_to_image(406            cast(DictionaryObject, self.indirect_reference.get_object()),407            pillow_parameters=kwargs,408        )409        assert extension is not None410        self.name = self.name[: self.name.rfind(".")] + extension411        self.data = byte_stream412        self.image = img413 414    def __str__(self) -> str:415        return f"{self.__class__.__name__}(name={self.name}, data: {_human_readable_bytes(len(self.data))})"416 417    def __repr__(self) -> str:418        return self.__str__()[:-1] + f", hash: {hash(self.data)})"419 420 421class VirtualListImages(Sequence[ImageFile]):422    """423    Provides access to images referenced within a page.424    Only one copy will be returned if the usage is used on the same page multiple times.425    See :func:`PageObject.images` for more details.426    """427 428    def __init__(429        self,430        ids_function: Callable[[], list[Union[str, list[str]]]],431        get_function: Callable[[Union[str, list[str], tuple[str]]], ImageFile],432    ) -> None:433        self.ids_function = ids_function434        self.get_function = get_function435        self.current = -1436 437    def __len__(self) -> int:438        return len(self.ids_function())439 440    def keys(self) -> list[Union[str, list[str]]]:441        return self.ids_function()442 443    def items(self) -> list[tuple[Union[str, list[str]], ImageFile]]:444        return [(x, self[x]) for x in self.ids_function()]445 446    @overload447    def __getitem__(self, index: Union[int, str, list[str]]) -> ImageFile:448        ...449 450    @overload451    def __getitem__(self, index: slice) -> Sequence[ImageFile]:452        ...453 454    def __getitem__(455        self, index: Union[int, slice, str, list[str], tuple[str]]456    ) -> Union[ImageFile, Sequence[ImageFile]]:457        lst = self.ids_function()458        if isinstance(index, slice):459            indices = range(*index.indices(len(self)))460            lst = [lst[x] for x in indices]461            cls = type(self)462            return cls((lambda: lst), self.get_function)463        if isinstance(index, (str, list, tuple)):464            return self.get_function(index)465        if not isinstance(index, int):466            raise TypeError("Invalid sequence indices type")467        len_self = len(lst)468        if index < 0:469            # support negative indexes470            index += len_self471        if not (0 <= index < len_self):472            raise IndexError("Sequence index out of range")473        return self.get_function(lst[index])474 475    def __iter__(self) -> Iterator[ImageFile]:476        for i in range(len(self)):477            yield self[i]478 479    def __str__(self) -> str:480        p = [f"Image_{i}={n}" for i, n in enumerate(self.ids_function())]481        return f"[{', '.join(p)}]"482 483 484class PageObject(DictionaryObject):485    """486    PageObject represents a single page within a PDF file.487 488    Typically these objects will be created by accessing the489    :attr:`pages<pypdf.PdfReader.pages>` property of the490    :class:`PdfReader<pypdf.PdfReader>` class, but it is491    also possible to create an empty page with the492    :meth:`create_blank_page()<pypdf._page.PageObject.create_blank_page>` static method.493 494    Args:495        pdf: PDF file the page belongs to.496        indirect_reference: Stores the original indirect reference to497            this object in its source PDF498 499    """500 501    original_page: "PageObject"  # very local use in writer when appending502 503    def __init__(504        self,505        pdf: Optional[PdfCommonDocProtocol] = None,506        indirect_reference: Optional[IndirectObject] = None,507    ) -> None:508        DictionaryObject.__init__(self)509        self.pdf = pdf510        self.inline_images: Optional[dict[str, ImageFile]] = None511        self.indirect_reference = indirect_reference512        if not is_null_or_none(indirect_reference):513            assert indirect_reference is not None, "mypy"514            self.update(cast(DictionaryObject, indirect_reference.get_object()))515        self._font_width_maps: dict[str, tuple[dict[str, float], str, float]] = {}516 517    def hash_bin(self) -> int:518        """519        Used to detect modified object.520 521        Note: this function is overloaded to return the same results522        as a DictionaryObject.523 524        Returns:525            Hash considering type and value.526 527        """528        return hash(529            (DictionaryObject, tuple(((k, v.hash_bin()) for k, v in self.items())))530        )531 532    def hash_value_data(self) -> bytes:533        data = super().hash_value_data()534        data += f"{id(self)}".encode()535        return data536 537    @property538    def user_unit(self) -> float:539        """540        A read-only positive number giving the size of user space units.541 542        It is in multiples of 1/72 inch. Hence a value of 1 means a user543        space unit is 1/72 inch, and a value of 3 means that a user544        space unit is 3/72 inch.545        """546        return cast(float, self.get(PG.USER_UNIT, 1))547 548    @staticmethod549    def create_blank_page(550        pdf: Optional[PdfCommonDocProtocol] = None,551        width: Union[float, Decimal, None] = None,552        height: Union[float, Decimal, None] = None,553    ) -> "PageObject":554        """555        Return a new blank page.556 557        If ``width`` or ``height`` is ``None``, try to get the page size558        from the last page of *pdf*.559 560        Args:561            pdf: PDF file the page is within.562            width: The width of the new page expressed in default user563                space units.564            height: The height of the new page expressed in default user565                space units.566 567        Returns:568            The new blank page569 570        Raises:571            PageSizeNotDefinedError: if ``pdf`` is ``None`` or contains572                no page573 574        """575        page = PageObject(pdf)576 577        # Creates a new page (cf PDF Reference §7.7.3.3)578        page.__setitem__(NameObject(PG.TYPE), NameObject("/Page"))579        page.__setitem__(NameObject(PG.PARENT), NullObject())580        page.__setitem__(NameObject(PG.RESOURCES), DictionaryObject())581        if width is None or height is None:582            if pdf is not None and len(pdf.pages) > 0:583                lastpage = pdf.pages[len(pdf.pages) - 1]584                width = lastpage.mediabox.width585                height = lastpage.mediabox.height586            else:587                raise PageSizeNotDefinedError588        page.__setitem__(589            NameObject(PG.MEDIABOX), RectangleObject((0, 0, width, height))  # type: ignore590        )591 592        return page593 594    def _get_ids_image(595        self,596        obj: Optional[DictionaryObject] = None,597        ancest: Optional[list[str]] = None,598        call_stack: Optional[list[Any]] = None,599    ) -> list[Union[str, list[str]]]:600        if call_stack is None:601            call_stack = []602        _i = getattr(obj, "indirect_reference", None)603        if _i in call_stack:604            return []605        call_stack.append(_i)606        if self.inline_images is None:607            self.inline_images = self._get_inline_images()608        if obj is None:609            obj = self610        if ancest is None:611            ancest = []612        lst: list[Union[str, list[str]]] = []613        if (614                PG.RESOURCES not in obj or615                is_null_or_none(resources := obj[PG.RESOURCES]) or616                RES.XOBJECT not in cast(DictionaryObject, resources)617        ):618            return [] if self.inline_images is None else list(self.inline_images.keys())619 620        x_object = resources[RES.XOBJECT].get_object()  # type: ignore621        for o in x_object:622            if not isinstance(x_object[o], StreamObject):623                continue624            if x_object[o][ImageAttributes.SUBTYPE] == "/Image":625                lst.append(o if len(ancest) == 0 else [*ancest, o])626            else:  # is a form with possible images inside627                lst.extend(self._get_ids_image(x_object[o], [*ancest, o], call_stack))628        assert self.inline_images is not None629        lst.extend(list(self.inline_images.keys()))630        return lst631 632    def _get_image(633        self,634        id: Union[str, list[str], tuple[str]],635        obj: Optional[DictionaryObject] = None,636    ) -> ImageFile:637        if obj is None:638            obj = cast(DictionaryObject, self)639        if isinstance(id, tuple):640            id = list(id)641        if isinstance(id, list) and len(id) == 1:642            id = id[0]643        xobjs: Optional[DictionaryObject] = None644        try:645            xobjs = cast(646                DictionaryObject, cast(DictionaryObject, obj[PG.RESOURCES])[RES.XOBJECT]647            )648        except KeyError as exc:649            if not (id[0] == "~" and id[-1] == "~"):650                raise KeyError(651                    f"Cannot access image object {id} without XObject resources"652                ) from exc653        if isinstance(id, str):654            if id[0] == "~" and id[-1] == "~":655                if self.inline_images is None:656                    self.inline_images = self._get_inline_images()657                if self.inline_images is None:658                    raise KeyError("No inline image can be found")659                return self.inline_images[id]660 661            assert xobjs is not None662            from .generic._image_xobject import _xobj_to_image  # noqa: PLC0415663            imgd = _xobj_to_image(cast(DictionaryObject, xobjs[id]))664            extension, byte_stream = imgd[:2]665            return ImageFile(666                name=f"{id[1:]}{extension}",667                data=byte_stream,668                image=imgd[2],669                indirect_reference=xobjs[id].indirect_reference,670            )671        # in a subobject672        assert xobjs is not None673        ids = id[1:]674        return self._get_image(ids, cast(DictionaryObject, xobjs[id[0]]))675 676    @property677    def images(self) -> VirtualListImages:678        """679        Read-only property emulating a list of images on a page.680 681        Get a list of all images on the page. The key can be:682        - A string (for the top object)683        - A tuple (for images within XObject forms)684        - An integer685 686        Examples:687            * `reader.pages[0].images[0]`        # return first image688            * `reader.pages[0].images['/I0']`    # return image '/I0'689            * `reader.pages[0].images['/TP1','/Image1']` # return image '/Image1' within '/TP1' XObject form690            * `for img in reader.pages[0].images:` # loops through all objects691 692        images.keys() and images.items() can be used.693 694        The ImageFile has the following properties:695 696            * `.name` : name of the object697            * `.data` : bytes of the object698            * `.image` : PIL Image Object699            * `.indirect_reference` : object reference700 701        and the following methods:702            `.replace(new_image: PIL.Image.Image, **kwargs)` :703                replace the image in the pdf with the new image704                applying the saving parameters indicated (such as quality)705 706        Example usage:707 708            reader.pages[0].images[0].replace(Image.open("new_image.jpg"), quality=20)709 710        Inline images are extracted and named ~0~, ~1~, ..., with the711        indirect_reference set to None.712 713        """714        return VirtualListImages(self._get_ids_image, self._get_image)715 716    def _translate_value_inline_image(self, k: str, v: PdfObject) -> PdfObject:717        """Translate values used in inline image"""718        try:719            v = NameObject(_INLINE_IMAGE_VALUE_MAPPING[cast(str, v)])720        except (TypeError, KeyError):721            if isinstance(v, NameObject):722                # It is a custom name, thus we have to look in resources.723                # The only applicable case is for ColorSpace.724                try:725                    res = cast(DictionaryObject, self["/Resources"])["/ColorSpace"]726                    v = cast(DictionaryObject, res)[v]727                except KeyError:  # for res and v728                    raise PdfReadError(f"Cannot find resource entry {v} for {k}")729        return v730 731    def _get_inline_images(self) -> dict[str, ImageFile]:732        """Load inline images. Entries will be identified as `~1~`."""733        content = self.get_contents()734        if is_null_or_none(content):735            return {}736        imgs_data = []737        assert content is not None, "mypy"738        for param, ope in content.operations:739            if ope == b"INLINE IMAGE":740                imgs_data.append(741                    {"settings": param["settings"], "__streamdata__": param["data"]}742                )743            elif ope in (b"BI", b"EI", b"ID"):  # pragma: no cover744                raise PdfReadError(745                    f"{ope!r} operator met whereas not expected, "746                    "please share use case with pypdf dev team"747                )748        files = {}749        for num, ii in enumerate(imgs_data):750            init = {751                "__streamdata__": ii["__streamdata__"],752                "/Length": len(ii["__streamdata__"]),753            }754            for k, v in ii["settings"].items():755                if k in {"/Length", "/L"}:  # no length is expected756                    continue757                if isinstance(v, list):758                    v = ArrayObject(759                        [self._translate_value_inline_image(k, x) for x in v]760                    )761                else:762                    v = self._translate_value_inline_image(k, v)763                k = NameObject(_INLINE_IMAGE_KEY_MAPPING[k])764                if k not in init:765                    init[k] = v766            ii["object"] = EncodedStreamObject.initialize_from_dictionary(init)767            from .generic._image_xobject import _xobj_to_image  # noqa: PLC0415768            extension, byte_stream, img = _xobj_to_image(ii["object"])769            files[f"~{num}~"] = ImageFile(770                name=f"~{num}~{extension}",771                data=byte_stream,772                image=img,773                indirect_reference=None,774            )775        return files776 777    @property778    def rotation(self) -> int:779        """780        The visual rotation of the page.781 782        This number has to be a multiple of 90 degrees: 0, 90, 180, or 270 are783        valid values. This property does not affect ``/Contents``.784        """785        rotate_obj = self.get(PG.ROTATE, 0)786        return rotate_obj if isinstance(rotate_obj, int) else rotate_obj.get_object()787 788    @rotation.setter789    def rotation(self, r: float) -> None:790        self[NameObject(PG.ROTATE)] = NumberObject((((int(r) + 45) // 90) * 90) % 360)791 792    def transfer_rotation_to_content(self) -> None:793        """794        Apply the rotation of the page to the content and the media/crop/...795        boxes.796 797        It is recommended to apply this function before page merging.798        """799        r = -self.rotation  # rotation to apply is in the otherway800        self.rotation = 0801        mb = RectangleObject(self.mediabox)802        trsf = (803            Transformation()804            .translate(805                -float(mb.left + mb.width / 2), -float(mb.bottom + mb.height / 2)806            )807            .rotate(r)808        )809        pt1 = trsf.apply_on(mb.lower_left)810        pt2 = trsf.apply_on(mb.upper_right)811        trsf = trsf.translate(-min(pt1[0], pt2[0]), -min(pt1[1], pt2[1]))812        self.add_transformation(trsf, False)813        for b in ["/MediaBox", "/CropBox", "/BleedBox", "/TrimBox", "/ArtBox"]:814            if b in self:815                rr = RectangleObject(self[b])  # type: ignore816                pt1 = trsf.apply_on(rr.lower_left)817                pt2 = trsf.apply_on(rr.upper_right)818                self[NameObject(b)] = RectangleObject(819                    (820                        min(pt1[0], pt2[0]),821                        min(pt1[1], pt2[1]),822                        max(pt1[0], pt2[0]),823                        max(pt1[1], pt2[1]),824                    )825                )826 827    def rotate(self, angle: int) -> "PageObject":828        """829        Rotate a page clockwise by increments of 90 degrees.830 831        Args:832            angle: Angle to rotate the page. Must be an increment of 90 deg.833 834        Returns:835            The rotated PageObject836 837        """838        if angle % 90 != 0:839            raise ValueError("Rotation angle must be a multiple of 90")840        self[NameObject(PG.ROTATE)] = NumberObject(self.rotation + angle)841        return self842 843    def _merge_resources(844        self,845        res1: DictionaryObject,846        res2: DictionaryObject,847        resource: Any,848        new_res1: bool = True,849    ) -> tuple[dict[str, Any], dict[str, Any]]:850        try:851            assert isinstance(self.indirect_reference, IndirectObject)852            pdf = self.indirect_reference.pdf853            is_pdf_writer = hasattr(854                pdf, "_add_object"855            )  # expect isinstance(pdf, PdfWriter)856        except (AssertionError, AttributeError):857            pdf = None858            is_pdf_writer = False859 860        def compute_unique_key(base_key: str) -> tuple[str, bool]:861            """862            Find a key that either doesn't already exist or has the same value863            (indicated by the bool)864 865            Args:866                base_key: An index is added to this to get the computed key867 868            Returns:869                A tuple (computed key, bool) where the boolean indicates870                if there is a resource of the given computed_key with the same871                value.872 873            """874            value = page2res.raw_get(base_key)875            # TODO: a possible improvement for writer, the indirect_reference876            # cannot be found because translated877 878            # try the current key first (e.g. "foo"), but otherwise iterate879            # through "foo-0", "foo-1", etc. new_res can contain only finitely880            # many keys, thus this'll eventually end, even if it's been crafted881            # to be maximally annoying.882            computed_key = base_key883            idx = 0884            while computed_key in new_res:885                if new_res.raw_get(computed_key) == value:886                    # there's already a resource of this name, with the exact887                    # same value888                    return computed_key, True889                computed_key = f"{base_key}-{idx}"890                idx += 1891            return computed_key, False892 893        if new_res1:894            new_res = DictionaryObject()895            new_res.update(res1.get(resource, DictionaryObject()).get_object())896        else:897            new_res = cast(DictionaryObject, res1[resource])898        page2res = cast(899            DictionaryObject, res2.get(resource, DictionaryObject()).get_object()900        )901        rename_res = {}902        for key in page2res:903            unique_key, same_value = compute_unique_key(key)904            newname = NameObject(unique_key)905            if key != unique_key:906                # we have to use a different name for this907                rename_res[key] = newname908 909            if not same_value:910                if is_pdf_writer:911                    new_res[newname] = page2res.raw_get(key).clone(pdf)912                    try:913                        new_res[newname] = new_res[newname].indirect_reference914                    except AttributeError:915                        pass916                else:917                    new_res[newname] = page2res.raw_get(key)918            lst = sorted(new_res.items())919            new_res.clear()920            for el in lst:921                new_res[el[0]] = el[1]922        return new_res, rename_res923 924    @staticmethod925    def _content_stream_rename(926        stream: ContentStream,927        rename: dict[Any, Any],928        pdf: Optional[PdfCommonDocProtocol],929    ) -> ContentStream:930        if not rename:931            return stream932        stream = ContentStream(stream, pdf)933        for operands, _operator in stream.operations:934            if isinstance(operands, list):935                for i, op in enumerate(operands):936                    if isinstance(op, NameObject):937                        operands[i] = rename.get(op, op)938            elif isinstance(operands, dict):939                for i, op in operands.items():940                    if isinstance(op, NameObject):941                        operands[i] = rename.get(op, op)942            else:943                raise KeyError(f"Type of operands is {type(operands)}")944        return stream945 946    @staticmethod947    def _add_transformation_matrix(948        contents: Any,949        pdf: Optional[PdfCommonDocProtocol],950        ctm: CompressedTransformationMatrix,951    ) -> ContentStream:952        """Add transformation matrix at the beginning of the given contents stream."""953        content_stream = ContentStream(contents, pdf)954        content_stream.operations.insert(955            0,956            (957                [FloatObject(x) for x in ctm],958                b"cm",959            ),960        )961        return content_stream962 963    def _get_contents_as_bytes(self) -> Optional[bytes]:964        """965        Return the page contents as bytes.966 967        Returns:968            The ``/Contents`` object as bytes, or ``None`` if it doesn't exist.969 970        """971        if PG.CONTENTS in self:972            obj = self[PG.CONTENTS].get_object()973            if isinstance(obj, list):974                return b"".join(x.get_object().get_data() for x in obj)975            return cast(EncodedStreamObject, obj).get_data()976        return None977 978    def get_contents(self) -> Optional[ContentStream]:979        """980        Access the page contents.981 982        Returns:983            The ``/Contents`` object, or ``None`` if it does not exist.984            ``/Contents`` is optional, as described in §7.7.3.3 of the PDF Reference.985 986        """987        if PG.CONTENTS in self:988            try:989                pdf = cast(IndirectObject, self.indirect_reference).pdf990            except AttributeError:991                pdf = None992            obj = self[PG.CONTENTS]993            if is_null_or_none(obj):994                return None995            resolved_object = obj.get_object()996            return ContentStream(resolved_object, pdf)997        return None998 999    def replace_contents(1000        self, content: Union[None, ContentStream, EncodedStreamObject, ArrayObject]1001    ) -> None:1002        """1003        Replace the page contents with the new content and nullify old objects1004        Args:1005            content: new content; if None delete the content field.1006        """1007        if not hasattr(self, "indirect_reference") or self.indirect_reference is None:1008            # the page is not attached : the content is directly attached.1009            self[NameObject(PG.CONTENTS)] = content1010            return1011 1012        from pypdf._writer import PdfWriter  # noqa: PLC04151013        if not isinstance(self.indirect_reference.pdf, PdfWriter):1014            deprecate(1015                "Calling `PageObject.replace_contents()` for pages not assigned to a writer is deprecated "1016                "and will be removed in pypdf 7.0.0. Attach the page to the writer first or use "1017                "`PdfWriter(clone_from=...)` directly. The existing approach has proved being unreliable."1018            )1019 1020        writer = self.indirect_reference.pdf1021        if isinstance(self.get(PG.CONTENTS, None), ArrayObject):1022            content_array = cast(ArrayObject, self[PG.CONTENTS])1023            for reference in content_array:1024                try:1025                    writer._replace_object(indirect_reference=reference.indirect_reference, obj=NullObject())1026                except ValueError:1027                    # Occurs when called on PdfReader.1028                    pass1029 1030        if isinstance(content, ArrayObject):1031            content = ArrayObject(writer._add_object(obj) for obj in content)1032 1033        if is_null_or_none(content):1034            if PG.CONTENTS not in self:1035                return1036            assert self[PG.CONTENTS].indirect_reference is not None1037            writer._replace_object(indirect_reference=self[PG.CONTENTS].indirect_reference, obj=NullObject())1038            del self[PG.CONTENTS]1039        elif not hasattr(self.get(PG.CONTENTS, None), "indirect_reference"):1040            try:1041                self[NameObject(PG.CONTENTS)] = writer._add_object(content)1042            except AttributeError:1043                # applies at least for page not in writer1044                # as a backup solution, we put content as an object although not in accordance with pdf ref1045                # this will be fixed with the _add_object1046                self[NameObject(PG.CONTENTS)] = content1047        else:1048            assert content is not None, "mypy"1049            content.indirect_reference = self[1050                PG.CONTENTS1051            ].indirect_reference  # TODO: in the future may require generation management1052            try:1053                writer._replace_object(indirect_reference=content.indirect_reference, obj=content)1054            except AttributeError:1055                # applies at least for page not in writer1056                # as a backup solution, we put content as an object although not in accordance with pdf ref1057                # this will be fixed with the _add_object1058                self[NameObject(PG.CONTENTS)] = content1059        # forces recalculation of inline_images1060        self.inline_images = None1061 1062    def merge_page(1063        self, page2: "PageObject", expand: bool = False, over: bool = True1064    ) -> None:1065        """1066        Merge the content streams of two pages into one.1067 1068        Resource references (e.g. fonts) are maintained from both pages.1069        The mediabox, cropbox, etc of this page are not altered.1070        The parameter page's content stream will1071        be added to the end of this page's content stream,1072        meaning that it will be drawn after, or "on top" of this page.1073 1074        Args:1075            page2: The page to be merged into this one. Should be1076                an instance of :class:`PageObject<PageObject>`.1077            over: set the page2 content over page1 if True (default) else under1078            expand: If True, the current page dimensions will be1079                expanded to accommodate the dimensions of the page to be merged.1080 1081        """1082        self._merge_page(page2, over=over, expand=expand)1083 1084    def _merge_page(1085        self,1086        page2: "PageObject",1087        page2_transformation: Optional[Callable[[Any], ContentStream]] = None,1088        ctm: Optional[CompressedTransformationMatrix] = None,1089        over: bool = True,1090        expand: bool = False,1091    ) -> None:1092        # First we work on merging the resource dictionaries. This allows us1093        # to find out what symbols in the content streams we might need to1094        # rename.1095        try:1096            assert isinstance(self.indirect_reference, IndirectObject)1097            if hasattr(self.indirect_reference.pdf, "_add_object"):  # to detect PdfWriter1098                return self._merge_page_writer(1099                    page2, page2_transformation, ctm, over, expand1100                )1101        except (AssertionError, AttributeError):1102            pass1103 1104        new_resources = DictionaryObject()1105        rename: dict[str, Any] = {}1106        original_resources = cast(DictionaryObject, self.get(PG.RESOURCES, DictionaryObject()).get_object())1107        page2_resources = cast(DictionaryObject, page2.get(PG.RESOURCES, DictionaryObject()).get_object())1108        new_annots = ArrayObject()1109 1110        for page in (self, page2):1111            if PG.ANNOTS in page:1112                annots = page[PG.ANNOTS]1113                if isinstance(annots, ArrayObject):1114                    new_annots.extend(annots)1115        self[NameObject(PG.ANNOTS)] = new_annots1116 1117        for res in (1118            RES.EXT_G_STATE,1119            RES.COLOR_SPACE,1120            RES.PATTERN,1121            RES.SHADING,1122            RES.XOBJECT,1123            RES.FONT,1124            RES.PROPERTIES,1125        ):1126            new, new_resource_name = self._merge_resources(1127                original_resources, page2_resources, res1128            )1129            if new:1130                new_resources[NameObject(res)] = new1131                rename.update(new_resource_name)1132 1133        # Combine /ProcSet sets, making sure there is a consistent order1134        new_resources[NameObject(RES.PROC_SET)] = ArrayObject(1135            sorted(1136                set(1137                    original_resources.get(RES.PROC_SET, ArrayObject()).get_object()1138                ).union(1139                    set(page2_resources.get(RES.PROC_SET, ArrayObject()).get_object())1140                )1141            )1142        )1143 1144        new_content_array = ArrayObject()1145        original_content = self.get_contents()1146        if original_content is not None:1147            original_content.isolate_graphics_state()1148            new_content_array.append(original_content)1149 1150        page2_content = page2.get_contents()1151        if page2_content is not None:1152            rect = getattr(page2, MERGE_CROP_BOX)1153            page2_content.operations.insert(1154                0,1155                (1156                    map(1157                        FloatObject,1158                        [1159                            rect.left,1160                            rect.bottom,1161                            rect.width,1162                            rect.height,1163                        ],1164                    ),1165                    b"re",1166                ),1167            )1168            page2_content.operations.insert(1, ([], b"W"))1169            page2_content.operations.insert(2, ([], b"n"))1170            if page2_transformation is not None:1171                page2_content = page2_transformation(page2_content)1172            page2_content = PageObject._content_stream_rename(1173                page2_content, rename, self.pdf1174            )1175            page2_content.isolate_graphics_state()1176            if over:1177                new_content_array.append(page2_content)1178            else:1179                new_content_array.insert(0, page2_content)1180 1181        # if expanding the page to fit a new page, calculate the new media box size1182        if expand:1183            self._expand_mediabox(page2, ctm)1184 1185        self.replace_contents(ContentStream(new_content_array, self.pdf))1186        self[NameObject(PG.RESOURCES)] = new_resources1187 1188        return None1189 1190    def _merge_page_writer(1191        self,1192        page2: "PageObject",1193        page2transformation: Optional[Callable[[Any], ContentStream]] = None,1194        ctm: Optional[CompressedTransformationMatrix] = None,1195        over: bool = True,1196        expand: bool = False,1197    ) -> None:1198        # First we work on merging the resource dictionaries. This allows us1199        # to find which symbols in the content streams we might need to1200        # rename.

Showing the first 1,200 of 2357 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai