codekingpro/portable-devtools
114k
1# Copyright (c) 2006, Mathieu Fenniak2# Copyright (c) 2007, Ashish Kulkarni <kulkarni.ashish@gmail.com>3#4# All rights reserved.5#6# Redistribution and use in source and binary forms, with or without7# modification, are permitted provided that the following conditions are8# met:9#10# * Redistributions of source code must retain the above copyright notice,11# this list of conditions and the following disclaimer.12# * Redistributions in binary form must reproduce the above copyright notice,13# this list of conditions and the following disclaimer in the documentation14# and/or other materials provided with the distribution.15# * The name of the author may not be used to endorse or promote products16# derived from this software without specific prior written permission.17#18# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"19# AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE20# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE21# ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE22# LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR23# CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF24# SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS25# INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN26# CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)27# ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE28# POSSIBILITY OF SUCH DAMAGE.29 30import math31from collections.abc import Iterable, Iterator, Sequence32from copy import deepcopy33from dataclasses import asdict, dataclass34from decimal import Decimal35from io import BytesIO36from pathlib import Path37from typing import (38 Any,39 Callable,40 Literal,41 Optional,42 Union,43 cast,44 overload,45)46 47from ._font import Font48from ._protocols import PdfCommonDocProtocol49from ._text_extraction import (50 _layout_mode,51)52from ._text_extraction._text_extractor import TextExtraction53from ._utils import (54 CompressedTransformationMatrix,55 TransformationMatrixType,56 _human_readable_bytes,57 deprecate,58 logger_warning,59 matrix_multiply,60)61from .constants import (62 _INLINE_IMAGE_KEY_MAPPING,63 _INLINE_IMAGE_VALUE_MAPPING,64 AnnotationDictionaryAttributes,65 ImageAttributes,66)67from .constants import PageAttributes as PG68from .constants import Resources as RES69from .errors import PageSizeNotDefinedError, PdfReadError70from .generic import (71 ArrayObject,72 ContentStream,73 DictionaryObject,74 EncodedStreamObject,75 FloatObject,76 IndirectObject,77 NameObject,78 NullObject,79 NumberObject,80 PdfObject,81 RectangleObject,82 StreamObject,83 is_null_or_none,84)85 86try:87 from PIL.Image import Image88 89 pil_not_imported = False90except ImportError:91 Image = object # type: ignore[assignment,misc,unused-ignore] # TODO: Remove unused-ignore on Python 3.1092 pil_not_imported = True # error will be raised only when using images93 94MERGE_CROP_BOX = "cropbox" # pypdf <= 3.4.0 used "trimbox"95 96 97def _get_rectangle(self: Any, name: str, defaults: Iterable[str]) -> RectangleObject:98 retval: Union[None, RectangleObject, ArrayObject, IndirectObject] = self.get(name)99 if isinstance(retval, RectangleObject):100 return retval101 if is_null_or_none(retval):102 for d in defaults:103 retval = self.get(d)104 if retval is not None:105 break106 if isinstance(retval, IndirectObject):107 retval = self.pdf.get_object(retval)108 if isinstance(retval, ArrayObject) and (length := len(retval)) > 4:109 logger_warning(f"Expected four values, got {length}: {retval}", __name__)110 retval = RectangleObject(tuple(retval[:4]))111 else:112 retval = RectangleObject(retval) # type: ignore113 _set_rectangle(self, name, retval)114 return retval115 116 117def _set_rectangle(self: Any, name: str, value: Union[RectangleObject, float]) -> None:118 self[NameObject(name)] = value119 120 121def _delete_rectangle(self: Any, name: str) -> None:122 del self[name]123 124 125def _create_rectangle_accessor(name: str, fallback: Iterable[str]) -> property:126 return property(127 lambda self: _get_rectangle(self, name, fallback),128 lambda self, value: _set_rectangle(self, name, value),129 lambda self: _delete_rectangle(self, name),130 )131 132 133class Transformation:134 """135 Represent a 2D transformation.136 137 The transformation between two coordinate systems is represented by a 3-by-3138 transformation matrix with the following form::139 140 a b 0141 c d 0142 e f 1143 144 Because a transformation matrix has only six elements that can be changed,145 it is usually specified in PDF as the six-element array [ a b c d e f ].146 147 Coordinate transformations are expressed as matrix multiplications::148 149 a b 0150 [ x′ y′ 1 ] = [ x y 1 ] × c d 0151 e f 1152 153 154 Example:155 >>> from pypdf import PdfWriter, Transformation156 >>> page = PdfWriter().add_blank_page(800, 600)157 >>> op = Transformation().scale(sx=2, sy=3).translate(tx=10, ty=20)158 >>> page.add_transformation(op)159 160 """161 162 def __init__(self, ctm: CompressedTransformationMatrix = (1, 0, 0, 1, 0, 0)) -> None:163 self.ctm = ctm164 165 @property166 def matrix(self) -> TransformationMatrixType:167 """168 Return the transformation matrix as a tuple of tuples in the form:169 170 ((a, b, 0), (c, d, 0), (e, f, 1))171 """172 return (173 (self.ctm[0], self.ctm[1], 0),174 (self.ctm[2], self.ctm[3], 0),175 (self.ctm[4], self.ctm[5], 1),176 )177 178 @staticmethod179 def compress(matrix: TransformationMatrixType) -> CompressedTransformationMatrix:180 """181 Compresses the transformation matrix into a tuple of (a, b, c, d, e, f).182 183 Args:184 matrix: The transformation matrix as a tuple of tuples.185 186 Returns:187 A tuple representing the transformation matrix as (a, b, c, d, e, f)188 189 """190 return (191 matrix[0][0],192 matrix[0][1],193 matrix[1][0],194 matrix[1][1],195 matrix[2][0],196 matrix[2][1],197 )198 199 def _to_cm(self) -> str:200 # Returns the cm operation string for the given transformation matrix201 return (202 f"{self.ctm[0]:.4f} {self.ctm[1]:.4f} {self.ctm[2]:.4f} "203 f"{self.ctm[3]:.4f} {self.ctm[4]:.4f} {self.ctm[5]:.4f} cm"204 )205 206 def transform(self, m: "Transformation") -> "Transformation":207 """208 Apply one transformation to another.209 210 Args:211 m: a Transformation to apply.212 213 Returns:214 A new ``Transformation`` instance215 216 Example:217 >>> from pypdf import PdfWriter, Transformation218 >>> height, width = 40, 50219 >>> page = PdfWriter().add_blank_page(800, 600)220 >>> op = Transformation((1, 0, 0, -1, 0, height)) # vertical mirror221 >>> op = Transformation().transform(Transformation((-1, 0, 0, 1, width, 0))) # horizontal mirror222 >>> page.add_transformation(op)223 224 """225 ctm = Transformation.compress(matrix_multiply(self.matrix, m.matrix))226 return Transformation(ctm)227 228 def translate(self, tx: float = 0, ty: float = 0) -> "Transformation":229 """230 Translate the contents of a page.231 232 Args:233 tx: The translation along the x-axis.234 ty: The translation along the y-axis.235 236 Returns:237 A new ``Transformation`` instance238 239 """240 m = self.ctm241 return Transformation(ctm=(m[0], m[1], m[2], m[3], m[4] + tx, m[5] + ty))242 243 def scale(244 self, sx: Optional[float] = None, sy: Optional[float] = None245 ) -> "Transformation":246 """247 Scale the contents of a page towards the origin of the coordinate system.248 249 Typically, that is the lower-left corner of the page. That can be250 changed by translating the contents / the page boxes.251 252 Args:253 sx: The scale factor along the x-axis.254 sy: The scale factor along the y-axis.255 256 Returns:257 A new Transformation instance with the scaled matrix.258 259 """260 if sx is None and sy is None:261 raise ValueError("Either sx or sy must be specified")262 if sx is None:263 sx = sy264 if sy is None:265 sy = sx266 assert sx is not None267 assert sy is not None268 op: TransformationMatrixType = ((sx, 0, 0), (0, sy, 0), (0, 0, 1))269 ctm = Transformation.compress(matrix_multiply(self.matrix, op))270 return Transformation(ctm)271 272 def rotate(self, rotation: float) -> "Transformation":273 """274 Rotate the contents of a page.275 276 Args:277 rotation: The angle of rotation in degrees.278 279 Returns:280 A new ``Transformation`` instance with the rotated matrix.281 282 """283 rotation = math.radians(rotation)284 op: TransformationMatrixType = (285 (math.cos(rotation), math.sin(rotation), 0),286 (-math.sin(rotation), math.cos(rotation), 0),287 (0, 0, 1),288 )289 ctm = Transformation.compress(matrix_multiply(self.matrix, op))290 return Transformation(ctm)291 292 def __repr__(self) -> str:293 return f"Transformation(ctm={self.ctm})"294 295 @overload296 def apply_on(self, pt: list[float], as_object: bool = False) -> list[float]:297 ...298 299 @overload300 def apply_on(301 self, pt: tuple[float, float], as_object: bool = False302 ) -> tuple[float, float]:303 ...304 305 def apply_on(306 self,307 pt: Union[tuple[float, float], list[float]],308 as_object: bool = False,309 ) -> Union[tuple[float, float], list[float]]:310 """311 Apply the transformation matrix on the given point.312 313 Args:314 pt: A tuple or list representing the point in the form (x, y).315 as_object: If True, return items as FloatObject, otherwise as plain floats.316 317 Returns:318 A tuple or list representing the transformed point in the form (x', y')319 320 """321 typ = FloatObject if as_object else float322 pt1 = (323 typ(float(pt[0]) * self.ctm[0] + float(pt[1]) * self.ctm[2] + self.ctm[4]),324 typ(float(pt[0]) * self.ctm[1] + float(pt[1]) * self.ctm[3] + self.ctm[5]),325 )326 return list(pt1) if isinstance(pt, list) else pt1327 328 329@dataclass330class ImageFile:331 """332 Image within the PDF file. *This object is not designed to be built.*333 334 This object should not be modified except using :func:`ImageFile.replace` to replace the image with a new one.335 """336 337 name: str = ""338 """339 Filename as identified within the PDF file.340 """341 342 data: bytes = b""343 """344 Data as bytes.345 """346 347 image: Optional[Image] = None348 """349 Data as PIL image.350 """351 352 indirect_reference: Optional[IndirectObject] = None353 """354 Reference to the object storing the stream.355 """356 357 def replace(self, new_image: Image, **kwargs: Any) -> None:358 """359 Replace the image with a new PIL image.360 361 Args:362 new_image (PIL.Image.Image): The new PIL image to replace the existing image.363 **kwargs: Additional keyword arguments to pass to `Image.save()`.364 365 Raises:366 TypeError: If the image is inline or in a PdfReader.367 TypeError: If the image does not belong to a PdfWriter.368 TypeError: If `new_image` is not a PIL Image.369 370 Note:371 This method replaces the existing image with a new image.372 It is not allowed for inline images or images within a PdfReader.373 The `kwargs` parameter allows passing additional parameters374 to `Image.save()`, such as quality.375 376 """377 if pil_not_imported:378 raise ImportError(379 "pillow is required to do image extraction. "380 "It can be installed via 'pip install pypdf[image]'"381 )382 383 from ._reader import PdfReader # noqa: PLC0415384 from .generic import DictionaryObject, PdfObject # noqa: PLC0415385 from .generic._image_xobject import _xobj_to_image # noqa: PLC0415386 387 if self.indirect_reference is None:388 raise TypeError("Cannot update an inline image.")389 if not hasattr(self.indirect_reference.pdf, "_id_translated"):390 raise TypeError("Cannot update an image not belonging to a PdfWriter.")391 if not isinstance(new_image, Image):392 raise TypeError("new_image shall be a PIL Image")393 b = BytesIO()394 new_image.save(b, "PDF", **kwargs)395 reader = PdfReader(b)396 page_image = reader.pages[0].images[0]397 assert page_image.indirect_reference is not None398 self.indirect_reference.pdf._objects[self.indirect_reference.idnum - 1] = (399 page_image.indirect_reference.get_object()400 )401 cast(402 PdfObject, self.indirect_reference.get_object()403 ).indirect_reference = self.indirect_reference404 # change the object attributes405 extension, byte_stream, img = _xobj_to_image(406 cast(DictionaryObject, self.indirect_reference.get_object()),407 pillow_parameters=kwargs,408 )409 assert extension is not None410 self.name = self.name[: self.name.rfind(".")] + extension411 self.data = byte_stream412 self.image = img413 414 def __str__(self) -> str:415 return f"{self.__class__.__name__}(name={self.name}, data: {_human_readable_bytes(len(self.data))})"416 417 def __repr__(self) -> str:418 return self.__str__()[:-1] + f", hash: {hash(self.data)})"419 420 421class VirtualListImages(Sequence[ImageFile]):422 """423 Provides access to images referenced within a page.424 Only one copy will be returned if the usage is used on the same page multiple times.425 See :func:`PageObject.images` for more details.426 """427 428 def __init__(429 self,430 ids_function: Callable[[], list[Union[str, list[str]]]],431 get_function: Callable[[Union[str, list[str], tuple[str]]], ImageFile],432 ) -> None:433 self.ids_function = ids_function434 self.get_function = get_function435 self.current = -1436 437 def __len__(self) -> int:438 return len(self.ids_function())439 440 def keys(self) -> list[Union[str, list[str]]]:441 return self.ids_function()442 443 def items(self) -> list[tuple[Union[str, list[str]], ImageFile]]:444 return [(x, self[x]) for x in self.ids_function()]445 446 @overload447 def __getitem__(self, index: Union[int, str, list[str]]) -> ImageFile:448 ...449 450 @overload451 def __getitem__(self, index: slice) -> Sequence[ImageFile]:452 ...453 454 def __getitem__(455 self, index: Union[int, slice, str, list[str], tuple[str]]456 ) -> Union[ImageFile, Sequence[ImageFile]]:457 lst = self.ids_function()458 if isinstance(index, slice):459 indices = range(*index.indices(len(self)))460 lst = [lst[x] for x in indices]461 cls = type(self)462 return cls((lambda: lst), self.get_function)463 if isinstance(index, (str, list, tuple)):464 return self.get_function(index)465 if not isinstance(index, int):466 raise TypeError("Invalid sequence indices type")467 len_self = len(lst)468 if index < 0:469 # support negative indexes470 index += len_self471 if not (0 <= index < len_self):472 raise IndexError("Sequence index out of range")473 return self.get_function(lst[index])474 475 def __iter__(self) -> Iterator[ImageFile]:476 for i in range(len(self)):477 yield self[i]478 479 def __str__(self) -> str:480 p = [f"Image_{i}={n}" for i, n in enumerate(self.ids_function())]481 return f"[{', '.join(p)}]"482 483 484class PageObject(DictionaryObject):485 """486 PageObject represents a single page within a PDF file.487 488 Typically these objects will be created by accessing the489 :attr:`pages<pypdf.PdfReader.pages>` property of the490 :class:`PdfReader<pypdf.PdfReader>` class, but it is491 also possible to create an empty page with the492 :meth:`create_blank_page()<pypdf._page.PageObject.create_blank_page>` static method.493 494 Args:495 pdf: PDF file the page belongs to.496 indirect_reference: Stores the original indirect reference to497 this object in its source PDF498 499 """500 501 original_page: "PageObject" # very local use in writer when appending502 503 def __init__(504 self,505 pdf: Optional[PdfCommonDocProtocol] = None,506 indirect_reference: Optional[IndirectObject] = None,507 ) -> None:508 DictionaryObject.__init__(self)509 self.pdf = pdf510 self.inline_images: Optional[dict[str, ImageFile]] = None511 self.indirect_reference = indirect_reference512 if not is_null_or_none(indirect_reference):513 assert indirect_reference is not None, "mypy"514 self.update(cast(DictionaryObject, indirect_reference.get_object()))515 self._font_width_maps: dict[str, tuple[dict[str, float], str, float]] = {}516 517 def hash_bin(self) -> int:518 """519 Used to detect modified object.520 521 Note: this function is overloaded to return the same results522 as a DictionaryObject.523 524 Returns:525 Hash considering type and value.526 527 """528 return hash(529 (DictionaryObject, tuple(((k, v.hash_bin()) for k, v in self.items())))530 )531 532 def hash_value_data(self) -> bytes:533 data = super().hash_value_data()534 data += f"{id(self)}".encode()535 return data536 537 @property538 def user_unit(self) -> float:539 """540 A read-only positive number giving the size of user space units.541 542 It is in multiples of 1/72 inch. Hence a value of 1 means a user543 space unit is 1/72 inch, and a value of 3 means that a user544 space unit is 3/72 inch.545 """546 return cast(float, self.get(PG.USER_UNIT, 1))547 548 @staticmethod549 def create_blank_page(550 pdf: Optional[PdfCommonDocProtocol] = None,551 width: Union[float, Decimal, None] = None,552 height: Union[float, Decimal, None] = None,553 ) -> "PageObject":554 """555 Return a new blank page.556 557 If ``width`` or ``height`` is ``None``, try to get the page size558 from the last page of *pdf*.559 560 Args:561 pdf: PDF file the page is within.562 width: The width of the new page expressed in default user563 space units.564 height: The height of the new page expressed in default user565 space units.566 567 Returns:568 The new blank page569 570 Raises:571 PageSizeNotDefinedError: if ``pdf`` is ``None`` or contains572 no page573 574 """575 page = PageObject(pdf)576 577 # Creates a new page (cf PDF Reference §7.7.3.3)578 page.__setitem__(NameObject(PG.TYPE), NameObject("/Page"))579 page.__setitem__(NameObject(PG.PARENT), NullObject())580 page.__setitem__(NameObject(PG.RESOURCES), DictionaryObject())581 if width is None or height is None:582 if pdf is not None and len(pdf.pages) > 0:583 lastpage = pdf.pages[len(pdf.pages) - 1]584 width = lastpage.mediabox.width585 height = lastpage.mediabox.height586 else:587 raise PageSizeNotDefinedError588 page.__setitem__(589 NameObject(PG.MEDIABOX), RectangleObject((0, 0, width, height)) # type: ignore590 )591 592 return page593 594 def _get_ids_image(595 self,596 obj: Optional[DictionaryObject] = None,597 ancest: Optional[list[str]] = None,598 call_stack: Optional[list[Any]] = None,599 ) -> list[Union[str, list[str]]]:600 if call_stack is None:601 call_stack = []602 _i = getattr(obj, "indirect_reference", None)603 if _i in call_stack:604 return []605 call_stack.append(_i)606 if self.inline_images is None:607 self.inline_images = self._get_inline_images()608 if obj is None:609 obj = self610 if ancest is None:611 ancest = []612 lst: list[Union[str, list[str]]] = []613 if (614 PG.RESOURCES not in obj or615 is_null_or_none(resources := obj[PG.RESOURCES]) or616 RES.XOBJECT not in cast(DictionaryObject, resources)617 ):618 return [] if self.inline_images is None else list(self.inline_images.keys())619 620 x_object = resources[RES.XOBJECT].get_object() # type: ignore621 for o in x_object:622 if not isinstance(x_object[o], StreamObject):623 continue624 if x_object[o][ImageAttributes.SUBTYPE] == "/Image":625 lst.append(o if len(ancest) == 0 else [*ancest, o])626 else: # is a form with possible images inside627 lst.extend(self._get_ids_image(x_object[o], [*ancest, o], call_stack))628 assert self.inline_images is not None629 lst.extend(list(self.inline_images.keys()))630 return lst631 632 def _get_image(633 self,634 id: Union[str, list[str], tuple[str]],635 obj: Optional[DictionaryObject] = None,636 ) -> ImageFile:637 if obj is None:638 obj = cast(DictionaryObject, self)639 if isinstance(id, tuple):640 id = list(id)641 if isinstance(id, list) and len(id) == 1:642 id = id[0]643 xobjs: Optional[DictionaryObject] = None644 try:645 xobjs = cast(646 DictionaryObject, cast(DictionaryObject, obj[PG.RESOURCES])[RES.XOBJECT]647 )648 except KeyError as exc:649 if not (id[0] == "~" and id[-1] == "~"):650 raise KeyError(651 f"Cannot access image object {id} without XObject resources"652 ) from exc653 if isinstance(id, str):654 if id[0] == "~" and id[-1] == "~":655 if self.inline_images is None:656 self.inline_images = self._get_inline_images()657 if self.inline_images is None:658 raise KeyError("No inline image can be found")659 return self.inline_images[id]660 661 assert xobjs is not None662 from .generic._image_xobject import _xobj_to_image # noqa: PLC0415663 imgd = _xobj_to_image(cast(DictionaryObject, xobjs[id]))664 extension, byte_stream = imgd[:2]665 return ImageFile(666 name=f"{id[1:]}{extension}",667 data=byte_stream,668 image=imgd[2],669 indirect_reference=xobjs[id].indirect_reference,670 )671 # in a subobject672 assert xobjs is not None673 ids = id[1:]674 return self._get_image(ids, cast(DictionaryObject, xobjs[id[0]]))675 676 @property677 def images(self) -> VirtualListImages:678 """679 Read-only property emulating a list of images on a page.680 681 Get a list of all images on the page. The key can be:682 - A string (for the top object)683 - A tuple (for images within XObject forms)684 - An integer685 686 Examples:687 * `reader.pages[0].images[0]` # return first image688 * `reader.pages[0].images['/I0']` # return image '/I0'689 * `reader.pages[0].images['/TP1','/Image1']` # return image '/Image1' within '/TP1' XObject form690 * `for img in reader.pages[0].images:` # loops through all objects691 692 images.keys() and images.items() can be used.693 694 The ImageFile has the following properties:695 696 * `.name` : name of the object697 * `.data` : bytes of the object698 * `.image` : PIL Image Object699 * `.indirect_reference` : object reference700 701 and the following methods:702 `.replace(new_image: PIL.Image.Image, **kwargs)` :703 replace the image in the pdf with the new image704 applying the saving parameters indicated (such as quality)705 706 Example usage:707 708 reader.pages[0].images[0].replace(Image.open("new_image.jpg"), quality=20)709 710 Inline images are extracted and named ~0~, ~1~, ..., with the711 indirect_reference set to None.712 713 """714 return VirtualListImages(self._get_ids_image, self._get_image)715 716 def _translate_value_inline_image(self, k: str, v: PdfObject) -> PdfObject:717 """Translate values used in inline image"""718 try:719 v = NameObject(_INLINE_IMAGE_VALUE_MAPPING[cast(str, v)])720 except (TypeError, KeyError):721 if isinstance(v, NameObject):722 # It is a custom name, thus we have to look in resources.723 # The only applicable case is for ColorSpace.724 try:725 res = cast(DictionaryObject, self["/Resources"])["/ColorSpace"]726 v = cast(DictionaryObject, res)[v]727 except KeyError: # for res and v728 raise PdfReadError(f"Cannot find resource entry {v} for {k}")729 return v730 731 def _get_inline_images(self) -> dict[str, ImageFile]:732 """Load inline images. Entries will be identified as `~1~`."""733 content = self.get_contents()734 if is_null_or_none(content):735 return {}736 imgs_data = []737 assert content is not None, "mypy"738 for param, ope in content.operations:739 if ope == b"INLINE IMAGE":740 imgs_data.append(741 {"settings": param["settings"], "__streamdata__": param["data"]}742 )743 elif ope in (b"BI", b"EI", b"ID"): # pragma: no cover744 raise PdfReadError(745 f"{ope!r} operator met whereas not expected, "746 "please share use case with pypdf dev team"747 )748 files = {}749 for num, ii in enumerate(imgs_data):750 init = {751 "__streamdata__": ii["__streamdata__"],752 "/Length": len(ii["__streamdata__"]),753 }754 for k, v in ii["settings"].items():755 if k in {"/Length", "/L"}: # no length is expected756 continue757 if isinstance(v, list):758 v = ArrayObject(759 [self._translate_value_inline_image(k, x) for x in v]760 )761 else:762 v = self._translate_value_inline_image(k, v)763 k = NameObject(_INLINE_IMAGE_KEY_MAPPING[k])764 if k not in init:765 init[k] = v766 ii["object"] = EncodedStreamObject.initialize_from_dictionary(init)767 from .generic._image_xobject import _xobj_to_image # noqa: PLC0415768 extension, byte_stream, img = _xobj_to_image(ii["object"])769 files[f"~{num}~"] = ImageFile(770 name=f"~{num}~{extension}",771 data=byte_stream,772 image=img,773 indirect_reference=None,774 )775 return files776 777 @property778 def rotation(self) -> int:779 """780 The visual rotation of the page.781 782 This number has to be a multiple of 90 degrees: 0, 90, 180, or 270 are783 valid values. This property does not affect ``/Contents``.784 """785 rotate_obj = self.get(PG.ROTATE, 0)786 return rotate_obj if isinstance(rotate_obj, int) else rotate_obj.get_object()787 788 @rotation.setter789 def rotation(self, r: float) -> None:790 self[NameObject(PG.ROTATE)] = NumberObject((((int(r) + 45) // 90) * 90) % 360)791 792 def transfer_rotation_to_content(self) -> None:793 """794 Apply the rotation of the page to the content and the media/crop/...795 boxes.796 797 It is recommended to apply this function before page merging.798 """799 r = -self.rotation # rotation to apply is in the otherway800 self.rotation = 0801 mb = RectangleObject(self.mediabox)802 trsf = (803 Transformation()804 .translate(805 -float(mb.left + mb.width / 2), -float(mb.bottom + mb.height / 2)806 )807 .rotate(r)808 )809 pt1 = trsf.apply_on(mb.lower_left)810 pt2 = trsf.apply_on(mb.upper_right)811 trsf = trsf.translate(-min(pt1[0], pt2[0]), -min(pt1[1], pt2[1]))812 self.add_transformation(trsf, False)813 for b in ["/MediaBox", "/CropBox", "/BleedBox", "/TrimBox", "/ArtBox"]:814 if b in self:815 rr = RectangleObject(self[b]) # type: ignore816 pt1 = trsf.apply_on(rr.lower_left)817 pt2 = trsf.apply_on(rr.upper_right)818 self[NameObject(b)] = RectangleObject(819 (820 min(pt1[0], pt2[0]),821 min(pt1[1], pt2[1]),822 max(pt1[0], pt2[0]),823 max(pt1[1], pt2[1]),824 )825 )826 827 def rotate(self, angle: int) -> "PageObject":828 """829 Rotate a page clockwise by increments of 90 degrees.830 831 Args:832 angle: Angle to rotate the page. Must be an increment of 90 deg.833 834 Returns:835 The rotated PageObject836 837 """838 if angle % 90 != 0:839 raise ValueError("Rotation angle must be a multiple of 90")840 self[NameObject(PG.ROTATE)] = NumberObject(self.rotation + angle)841 return self842 843 def _merge_resources(844 self,845 res1: DictionaryObject,846 res2: DictionaryObject,847 resource: Any,848 new_res1: bool = True,849 ) -> tuple[dict[str, Any], dict[str, Any]]:850 try:851 assert isinstance(self.indirect_reference, IndirectObject)852 pdf = self.indirect_reference.pdf853 is_pdf_writer = hasattr(854 pdf, "_add_object"855 ) # expect isinstance(pdf, PdfWriter)856 except (AssertionError, AttributeError):857 pdf = None858 is_pdf_writer = False859 860 def compute_unique_key(base_key: str) -> tuple[str, bool]:861 """862 Find a key that either doesn't already exist or has the same value863 (indicated by the bool)864 865 Args:866 base_key: An index is added to this to get the computed key867 868 Returns:869 A tuple (computed key, bool) where the boolean indicates870 if there is a resource of the given computed_key with the same871 value.872 873 """874 value = page2res.raw_get(base_key)875 # TODO: a possible improvement for writer, the indirect_reference876 # cannot be found because translated877 878 # try the current key first (e.g. "foo"), but otherwise iterate879 # through "foo-0", "foo-1", etc. new_res can contain only finitely880 # many keys, thus this'll eventually end, even if it's been crafted881 # to be maximally annoying.882 computed_key = base_key883 idx = 0884 while computed_key in new_res:885 if new_res.raw_get(computed_key) == value:886 # there's already a resource of this name, with the exact887 # same value888 return computed_key, True889 computed_key = f"{base_key}-{idx}"890 idx += 1891 return computed_key, False892 893 if new_res1:894 new_res = DictionaryObject()895 new_res.update(res1.get(resource, DictionaryObject()).get_object())896 else:897 new_res = cast(DictionaryObject, res1[resource])898 page2res = cast(899 DictionaryObject, res2.get(resource, DictionaryObject()).get_object()900 )901 rename_res = {}902 for key in page2res:903 unique_key, same_value = compute_unique_key(key)904 newname = NameObject(unique_key)905 if key != unique_key:906 # we have to use a different name for this907 rename_res[key] = newname908 909 if not same_value:910 if is_pdf_writer:911 new_res[newname] = page2res.raw_get(key).clone(pdf)912 try:913 new_res[newname] = new_res[newname].indirect_reference914 except AttributeError:915 pass916 else:917 new_res[newname] = page2res.raw_get(key)918 lst = sorted(new_res.items())919 new_res.clear()920 for el in lst:921 new_res[el[0]] = el[1]922 return new_res, rename_res923 924 @staticmethod925 def _content_stream_rename(926 stream: ContentStream,927 rename: dict[Any, Any],928 pdf: Optional[PdfCommonDocProtocol],929 ) -> ContentStream:930 if not rename:931 return stream932 stream = ContentStream(stream, pdf)933 for operands, _operator in stream.operations:934 if isinstance(operands, list):935 for i, op in enumerate(operands):936 if isinstance(op, NameObject):937 operands[i] = rename.get(op, op)938 elif isinstance(operands, dict):939 for i, op in operands.items():940 if isinstance(op, NameObject):941 operands[i] = rename.get(op, op)942 else:943 raise KeyError(f"Type of operands is {type(operands)}")944 return stream945 946 @staticmethod947 def _add_transformation_matrix(948 contents: Any,949 pdf: Optional[PdfCommonDocProtocol],950 ctm: CompressedTransformationMatrix,951 ) -> ContentStream:952 """Add transformation matrix at the beginning of the given contents stream."""953 content_stream = ContentStream(contents, pdf)954 content_stream.operations.insert(955 0,956 (957 [FloatObject(x) for x in ctm],958 b"cm",959 ),960 )961 return content_stream962 963 def _get_contents_as_bytes(self) -> Optional[bytes]:964 """965 Return the page contents as bytes.966 967 Returns:968 The ``/Contents`` object as bytes, or ``None`` if it doesn't exist.969 970 """971 if PG.CONTENTS in self:972 obj = self[PG.CONTENTS].get_object()973 if isinstance(obj, list):974 return b"".join(x.get_object().get_data() for x in obj)975 return cast(EncodedStreamObject, obj).get_data()976 return None977 978 def get_contents(self) -> Optional[ContentStream]:979 """980 Access the page contents.981 982 Returns:983 The ``/Contents`` object, or ``None`` if it does not exist.984 ``/Contents`` is optional, as described in §7.7.3.3 of the PDF Reference.985 986 """987 if PG.CONTENTS in self:988 try:989 pdf = cast(IndirectObject, self.indirect_reference).pdf990 except AttributeError:991 pdf = None992 obj = self[PG.CONTENTS]993 if is_null_or_none(obj):994 return None995 resolved_object = obj.get_object()996 return ContentStream(resolved_object, pdf)997 return None998 999 def replace_contents(1000 self, content: Union[None, ContentStream, EncodedStreamObject, ArrayObject]1001 ) -> None:1002 """1003 Replace the page contents with the new content and nullify old objects1004 Args:1005 content: new content; if None delete the content field.1006 """1007 if not hasattr(self, "indirect_reference") or self.indirect_reference is None:1008 # the page is not attached : the content is directly attached.1009 self[NameObject(PG.CONTENTS)] = content1010 return1011 1012 from pypdf._writer import PdfWriter # noqa: PLC04151013 if not isinstance(self.indirect_reference.pdf, PdfWriter):1014 deprecate(1015 "Calling `PageObject.replace_contents()` for pages not assigned to a writer is deprecated "1016 "and will be removed in pypdf 7.0.0. Attach the page to the writer first or use "1017 "`PdfWriter(clone_from=...)` directly. The existing approach has proved being unreliable."1018 )1019 1020 writer = self.indirect_reference.pdf1021 if isinstance(self.get(PG.CONTENTS, None), ArrayObject):1022 content_array = cast(ArrayObject, self[PG.CONTENTS])1023 for reference in content_array:1024 try:1025 writer._replace_object(indirect_reference=reference.indirect_reference, obj=NullObject())1026 except ValueError:1027 # Occurs when called on PdfReader.1028 pass1029 1030 if isinstance(content, ArrayObject):1031 content = ArrayObject(writer._add_object(obj) for obj in content)1032 1033 if is_null_or_none(content):1034 if PG.CONTENTS not in self:1035 return1036 assert self[PG.CONTENTS].indirect_reference is not None1037 writer._replace_object(indirect_reference=self[PG.CONTENTS].indirect_reference, obj=NullObject())1038 del self[PG.CONTENTS]1039 elif not hasattr(self.get(PG.CONTENTS, None), "indirect_reference"):1040 try:1041 self[NameObject(PG.CONTENTS)] = writer._add_object(content)1042 except AttributeError:1043 # applies at least for page not in writer1044 # as a backup solution, we put content as an object although not in accordance with pdf ref1045 # this will be fixed with the _add_object1046 self[NameObject(PG.CONTENTS)] = content1047 else:1048 assert content is not None, "mypy"1049 content.indirect_reference = self[1050 PG.CONTENTS1051 ].indirect_reference # TODO: in the future may require generation management1052 try:1053 writer._replace_object(indirect_reference=content.indirect_reference, obj=content)1054 except AttributeError:1055 # applies at least for page not in writer1056 # as a backup solution, we put content as an object although not in accordance with pdf ref1057 # this will be fixed with the _add_object1058 self[NameObject(PG.CONTENTS)] = content1059 # forces recalculation of inline_images1060 self.inline_images = None1061 1062 def merge_page(1063 self, page2: "PageObject", expand: bool = False, over: bool = True1064 ) -> None:1065 """1066 Merge the content streams of two pages into one.1067 1068 Resource references (e.g. fonts) are maintained from both pages.1069 The mediabox, cropbox, etc of this page are not altered.1070 The parameter page's content stream will1071 be added to the end of this page's content stream,1072 meaning that it will be drawn after, or "on top" of this page.1073 1074 Args:1075 page2: The page to be merged into this one. Should be1076 an instance of :class:`PageObject<PageObject>`.1077 over: set the page2 content over page1 if True (default) else under1078 expand: If True, the current page dimensions will be1079 expanded to accommodate the dimensions of the page to be merged.1080 1081 """1082 self._merge_page(page2, over=over, expand=expand)1083 1084 def _merge_page(1085 self,1086 page2: "PageObject",1087 page2_transformation: Optional[Callable[[Any], ContentStream]] = None,1088 ctm: Optional[CompressedTransformationMatrix] = None,1089 over: bool = True,1090 expand: bool = False,1091 ) -> None:1092 # First we work on merging the resource dictionaries. This allows us1093 # to find out what symbols in the content streams we might need to1094 # rename.1095 try:1096 assert isinstance(self.indirect_reference, IndirectObject)1097 if hasattr(self.indirect_reference.pdf, "_add_object"): # to detect PdfWriter1098 return self._merge_page_writer(1099 page2, page2_transformation, ctm, over, expand1100 )1101 except (AssertionError, AttributeError):1102 pass1103 1104 new_resources = DictionaryObject()1105 rename: dict[str, Any] = {}1106 original_resources = cast(DictionaryObject, self.get(PG.RESOURCES, DictionaryObject()).get_object())1107 page2_resources = cast(DictionaryObject, page2.get(PG.RESOURCES, DictionaryObject()).get_object())1108 new_annots = ArrayObject()1109 1110 for page in (self, page2):1111 if PG.ANNOTS in page:1112 annots = page[PG.ANNOTS]1113 if isinstance(annots, ArrayObject):1114 new_annots.extend(annots)1115 self[NameObject(PG.ANNOTS)] = new_annots1116 1117 for res in (1118 RES.EXT_G_STATE,1119 RES.COLOR_SPACE,1120 RES.PATTERN,1121 RES.SHADING,1122 RES.XOBJECT,1123 RES.FONT,1124 RES.PROPERTIES,1125 ):1126 new, new_resource_name = self._merge_resources(1127 original_resources, page2_resources, res1128 )1129 if new:1130 new_resources[NameObject(res)] = new1131 rename.update(new_resource_name)1132 1133 # Combine /ProcSet sets, making sure there is a consistent order1134 new_resources[NameObject(RES.PROC_SET)] = ArrayObject(1135 sorted(1136 set(1137 original_resources.get(RES.PROC_SET, ArrayObject()).get_object()1138 ).union(1139 set(page2_resources.get(RES.PROC_SET, ArrayObject()).get_object())1140 )1141 )1142 )1143 1144 new_content_array = ArrayObject()1145 original_content = self.get_contents()1146 if original_content is not None:1147 original_content.isolate_graphics_state()1148 new_content_array.append(original_content)1149 1150 page2_content = page2.get_contents()1151 if page2_content is not None:1152 rect = getattr(page2, MERGE_CROP_BOX)1153 page2_content.operations.insert(1154 0,1155 (1156 map(1157 FloatObject,1158 [1159 rect.left,1160 rect.bottom,1161 rect.width,1162 rect.height,1163 ],1164 ),1165 b"re",1166 ),1167 )1168 page2_content.operations.insert(1, ([], b"W"))1169 page2_content.operations.insert(2, ([], b"n"))1170 if page2_transformation is not None:1171 page2_content = page2_transformation(page2_content)1172 page2_content = PageObject._content_stream_rename(1173 page2_content, rename, self.pdf1174 )1175 page2_content.isolate_graphics_state()1176 if over:1177 new_content_array.append(page2_content)1178 else:1179 new_content_array.insert(0, page2_content)1180 1181 # if expanding the page to fit a new page, calculate the new media box size1182 if expand:1183 self._expand_mediabox(page2, ctm)1184 1185 self.replace_contents(ContentStream(new_content_array, self.pdf))1186 self[NameObject(PG.RESOURCES)] = new_resources1187 1188 return None1189 1190 def _merge_page_writer(1191 self,1192 page2: "PageObject",1193 page2transformation: Optional[Callable[[Any], ContentStream]] = None,1194 ctm: Optional[CompressedTransformationMatrix] = None,1195 over: bool = True,1196 expand: bool = False,1197 ) -> None:1198 # First we work on merging the resource dictionaries. This allows us1199 # to find which symbols in the content streams we might need to1200 # rename.