Team Ai
Apppublic

documentExtractionag051/ExtractDocument

sourceHugging Faceupdated 9mo agoView on Hugging Face
0likes
pdf_utils.cpython-313.pyc146 linesDownload Raw Back to __pycache__
1�

2�G_iJ���SrSSKrSSKrSSKJrJrJrJr SSKrSr	SSKJr Sr
SrSrS	\4S3jrS\\\4S\\\4S	\4S
jrS\S	\4SjrS\\\\\4S	\\\\\44SjrS\\\\4S\S\S	\\\\44SjrS.S\\\\4S\S\S	\\\\\44SjjrS\\\\\4S\S	\\\\44SjrS.S\\\\4S\S\S\S\S	\\\\44Sjjr\4S\S\S	\\\44Sjjr\4S \S!\S\S	\\\44S"jjrS#\S\S	\\\44S$jr\ S%:Xa�SSK!r!SSK"r"\#"\!RH5S&:a"\%"S'5 \%"S(5 \!RL"S)5 \!RHS)r'\#"\!RH5S&:�a\"\!RHS&5O\r(\%"S*\'35 \%"S+\(35 \%"S,5 \"\'\(5r)\%"\"RT"\)S&S-95 gg!\4a Sr	GN�f=f!\5a Sr
GN�f=f)/a�6PDF Text Extraction Utility Module7 8Extracts text and bounding box coordinates from digital/searchable PDFs.9Uses pdfplumber as primary library with pypdf as fallback.10Converts word-level extractions into properly formatted line segments.11 12Output JSON Schema:13{14    "version": "1.0",15    "metadata": {...},16    "pages": [{"pageNum": 1, "width": ..., "height": ...}],17    "ocrBlocks": [...]18}19�N)�Dict�List�Any�SetTF)�	PdfReader���Q��?i�	�returnc�Z�[[R"55R5$)z:Generate a unique uppercase UUID for block identification.)�str�uuid�uuid4�upper���aC:\Users\poona\OneDrive\Documents\nextlifyai\Document_Automation\Document_Extraction\pdf_utils.py�_generate_uuidr0s���t�z�z�|��"�"�$�$r�rect1�rect2c��[USUS5n[USUS5n[USUS5n[USUS5nXB:dXS:agXB-20XS-21-nUSUS-22USUS-23-nUS:Xag[Xg-S-5$)z�24Calculate the percentage of intersection between two rectangles.25 26Args:27    rect1: First rectangle with x1, y1, x2, y228    rect2: Second rectangle with x1, y1, x2, y229    30Returns:31    Intersection percentage (0-100)32�x1�y1�x2�y2r�d)�max�min�int)rr�x_left�y_top�x_right�y_bottom�intersection_area�33rect1_areas        r�_intersection_pctr$5s�����t��e�D�k�
*�F���d��U�4�[�)�E��%��+�u�T�{�+�G��5��;��d��,�H���8�+�� �)�h�.>�?����+��d��+��d��e�D�k�0I�J�J��Q����!�.�#�5�6�6r�pdf_pathc	��[(a�[R"U5nURS[	S[UR55H;nUR
5nU(dMUR5(dM3 SSS5 g SSS5 g[(ap[U5nURS[	S[UR55H3nUR
5nU(dMUR5(dM3 g gg!,(df   g=f![a N�f=f![a gf=f)z8Check if PDF has embedded text (searchable/digital PDF).N�TF)�PDFPLUMBER_AVAILABLE�34pdfplumber�open�pagesr�len�extract_text�strip�	Exception�PYPDF_AVAILABLEr)r%�pdf�page�text�readers     r�is_pdf_digitalr5Us����	�����*�c��I�I�&=�s�1�c�#�)�)�n�'=�>�D��,�,�.�D��t��35�36���#�	+�*�>�+�37���	��x�(�F����%?�c�!�S����->�&?�@���(�(�*���4�D�J�J�L�L��A���)+�*�38���	��	���	���	�sl�D%�AD�,D�D�D%�
D�D%�$AD5�7D5�D5�D5�39D"�D%�"D%�%40D2�1D2�541E�E�	all_linesc���[5n/n[U5H�up4X1;dMU(dMUR5nUSSSnUSSSn[U5H]up�U	(dMU	Sn42X8:wdMXjSS::dM'XzSS:�dM4X�;dM;URU5 UR	U	5 M_ URSS9 URU5 UR
U5 M� U$)z(Merge lines that may overlap vertically.������geometryrrc��USS$)Nr9rr)�blocks r�<lambda>�_merge_lines.<locals>.<lambda>�s��E�*�,=�d�,Cr��key)�set�	enumerate�copy�add�extend�sort�append)r6�used_ids�new_page_lines�i�	page_line�new_linerr�j�page_line_subsequent�43last_blocks           r�_merge_linesrOts�����H�13�N�!�)�,������� �~�~�'�H��2��z�*�4�0�B��2��z�*�4�0�B�+4�Y�+?�'��'�'�!5�b�!9�J�����4�T�:�:���4�T�:�:��)� ���Q�� ���(<�=�,@�
�M�M�C�M�D��L�L��O��!�!�(�+�%-�(�r�blocks�height�widthc��U(d/$SnSnSn[U5U:�a[USS9$/n[5nUGHYnUSU;dM/n	UGH&n44U45SU;dMUSSUSS	U46SS47U48SSS.n[USU5n[X�S5n
X�:�a,X�:�a'U	R	U495 URU50S5 M|USSUSS	-S
U--nUSS51USS-S
U--nU52SSU53SS	-S
U--nU54SS55U56SS-S
U--nUU-57S
-UU-58S
--S-nUS::dM�X�:�dX�:�dGMU	R	U595 URU60S5 GM) U	(dGMHUR	U	5 GM\ U(a[
U5nURSS9 /nUHn	URU	5 M U$)z;Sort words in reading order (top to bottom, left to right).i��A�Zc�"�USSUSS4$)Nr9rrr)�bs rr<�._sort_words_in_reading_order.<locals>.<lambda>�s��Q�z�]�4�-@�!�J�-�PT�BU�,Vrr>�idr9rrrr)rrrr�g�?g�������?c�*�U(aUSSS$S$)Nrr9rr)�lines rr<rX�s��T��Q��61�(;�D�(A�(P�q�(Pr)	r,�sortedr@r$rFrCrOrErD)rPrQrR�	max_words�y_axis_intersection_thresh�alternate_threshr6rG�62seed_blockr\r;�new_rect_y_axis�thresh1�thresh2�
seed_center_x�
seed_center_y�block_center_x�block_center_y�dist�all_sorted_liness                    r�_sort_words_in_reading_orderrk�sT����	��I�!#����63�6�{�i���f�"V�W�W�,.�I���H��64��d��8�+�)+�D�����;�h�.�(��4�T�:�(��4�T�:�#�J�/��5�#�J�/��5�	'�O�0�65�:�0F��X�G�/��J�AW�X�G��<��Af����E�*� ���U�4�[�1�)3�J�)?��)E�66�S]�H^�_c�Hd�)d�ij�mr�ir�(s�
�)3�J�)?��)E�67�S]�H^�_c�Hd�)d�ij�ms�is�(t�
�*/�68�*;�D�*A�E�*�DU�VZ�D[�*[�`a�di�`i�)j��*/�69�*;�D�*A�E�*�DU�VZ�D[�*[�`a�dj�`j�)k��!.��!?�!� C�}�We�Ge�hi�Fi� i�lo�o���3�;�G�,G�7�Kf� �K�K��.�$�L�L��t��5�3 �6�t�� � ��&�?�B� ��+�	����P��Q�-/��������%���r�line_segment_merge_thresholdc��[5n/nUGH+nURS5S:waMUSU;dM&U/nURUS5 UH�nURS5S:wdUSUS:Xd	USU;aM/USSUSSUSS-70S--nUSSUs=:aUSS:dMgO MkUSn	USS	U	SS71-72n73US:�aX�-OSnX�:dM�USS	U	SS74:�dM�URU5 URUS5 M� U(dGMURU5 GM. U$)z2Identify line segments as groups of related words.�	blockType�WORDrYr9rrrZr8rrr)r@�getrCrF)rPrRrl�added_blocks�
line_segmentsra�line_segmentr;�vert_mid_point�75dist_block�horizontal_gap�	gap_ratios            r�_identify_line_segmentsrx�s���!�U�L�02�M��76��>�>�+�&�&�0���d��<�/�&�<�L����Z��-�.����9�9�[�)�V�3�u�T�{�j�QU�FV�7V�Z_�`d�Ze�iu�Zu��"'�77�"3�D�"9�U�:�=N�t�=T�W\�]g�Wh�im�Wn�=n�rs�<s�"s���j�)�$�/�.�_�:�j�CY�Z^�C_�_�_�!-�b�!1�J�%*�:�%6�t�%<�z�*�?U�VZ�?[�%[�N�:?�!�)�� 6��I� �?�E�*�DU�VZ�D[�^h�is�^t�uy�^z�Dz�$�+�+�E�2�$�(�(��t��5� ��|��$�$�\�2�1�4�rrr�page_numc
��/nUH�nU(dMSRUVs/sHoDSPM	 sn5R5nU(dMGUVs/sHoDRSS5PM nnU(a[U5[	U5-OSn[SU55n[SU55n	[
SU55n78[
SU55nUR[5US	UUX�X�S79.S.5 M� U$s snfs snf)z&Format line segments into line blocks.� r3�80confidence��?c3�0# �UHoSSv� M g7f)r9rNr��.0r;s  r�	<genexpr>�(_format_line_segments.<locals>.<genexpr>����C�l�U�z�"�4�(�l���c3�0# �UHoSSv� M g7f)r9rNrrs  rr�r�r�r�c3�0# �UHoSSv� M g7f)r9rNrrs  rr�r�	r�r�c3�0# �UHoSSv� M g7f)r9rNrrs  rr�r�81r�r��LINE�rrrr�rY�pageNumrnr3r|r9)	�joinr.rp�sumr,rrrFr)rrry�formatted_segmentsrsr;�segment_text�confidencesr|rrrrs            r�_format_line_segmentsr��s���8202��%�����x�x�L� I�L�5�v��L� I�J�P�P�R����AM�N���y�y��s�3���N�<G�S��%��K�(8�8�S�83�
�C�l�C�
C��
�C�l�C�
C��
�C�l�C�
C��
�C�l�C�
C���!�!� �"��� �$�!�2�@�
#84�	�!&�2���+!J��Os�C>85�D�word_blocks�86page_width�page_heightc��U(d/$UHnSU;dM[5US'M [XU5n[XaU5n[Xs5nU$)zBPost-process word blocks to create properly separated line blocks.rY)rrkrxr�)	r�r�r�ryrlr;�
sorted_blocksrr�line_blockss	         r�_post_process_words_to_linesr�sU����	����u��(�*�E�$�K��1��:�V�M�+�M�Gc�d�M�'�
�@�K��rc���SSKnURRU5n[U5(d[	X15$[87(d[
S5eSSUSSUS.//S.n[R"U5n[UR5US	S88'[URSS9GH,upgURS:�a[UR-OSn[URU-5n	[URU-5n89US
R!UU	U90S.5 UR#SSSS9n/nUH}n
[%5USU
SS[U
SU-5[U
SU-5[U
SU-5[U
SU-5S.S.nUR!U5 USR!U5 M ['UU	U91UU5nUSR)U5 GM/ SSS5 U$!,(df   U$=f)z�92Main entry point for PDF text extraction.93 94Args:95    pdf_path: Path to the PDF file96    line_segment_merge_threshold: Gap threshold for line merging (default: 0.03)97    98Returns:99    Structured output dict with metadata and pages containing ocrBlocks100rN�<pdfplumber is required. Install with: pip install pdfplumber�1.0�
pdf-extractedr)��101documentId�documentName�source�
numberOfPages�lineSegmentMergeThreshold��version�metadatar+�	ocrBlocksr�r����startr+�r�rRrQFrZ��keep_blank_chars�x_tolerance�y_toleranceror3r}�x0�topr�bottomr�r�r�)�os�path�basenamer5�_create_empty_outputr(�RuntimeErrorr)r*r,r+rArR�DEFAULT_SCALE_WIDTHrrQrF�
extract_wordsrr�rD)r%rlr��filename�resultr1ryr2�scale_factorr�r��wordsr��word�102word_blockr�s                r�extract_text_from_pdfr�2s����w�w����)�H��(�#�#�#�H�K�K����Y�Z�Z��)�$�"��)E�103����F�104����	"�c�.1�#�)�)�n��z��?�+�'��	�	��;�N�H�?C�z�z�A�~�.����;�ST�L��T�Z�Z�,�6�7�J��d�k�k�L�8�9�K�
�7�O�"�"�#�#�%�$�
��&�&�!&���'��E�13�K���(�*�'�!'� ��L�"%�!�$�t�*�|�";�<�!�$�u�+��"<�=�!�$�t�*�|�";�<�!�$�x�.�<�"?�@�	!�
�105��"�"�:�.��{�#�*�*�:�6��$7�����,��K�
�;��&�&�{�3�c<�106#�l�M�m107#�	"�l�M�s
�4E!G�108G.�	pdf_bytes�	file_namec��[(d[S5eSSUSSUS.//S.n[R"U5n[R109"U5n[
UR5USS	'[URS110S9GH,upgURS:�a[UR-OS111n[URU-5n	[URU-5n112USRUU	U113S
.5 URSSSS9n/nUH}n
[5USU
SS[U
SU-5[U
SU-5[U
SU-5[U
SU-5S.S.nURU5 USRU5 M [!UU	U114UU5nUSR#U5 GM/ SSS5 U$!,(df   U$=f)a1115Extract text from in-memory PDF bytes (for file uploads).116 117Args:118    pdf_bytes: PDF file content as bytes119    file_name: Name of the file (for metadata)120    line_segment_merge_threshold: Gap threshold for line merging121    122Returns:123    Structured output dict with metadata and pages containing ocrBlocks124r�r�r�r)rr�r�r�r�r�r�r+r�FrZr�ror3r}r�r�rr�r�r�r�N)r(r��io�BytesIOr)r*r,r+rArRr�rrQrFr�rr�rD)r�r�rlr��125pdf_streamr1ryr2r�r�r�r�r�r�r�r�s                r�extract_text_from_pdf_bytesr��s���  ���Y�Z�Z��)�%�"��)E�126����F����I�&�J�	����	$��.1�#�)�)�n��z��?�+�'��	�	��;�N�H�?C�z�z�A�~�.����;�ST�L��T�Z�Z�,�6�7�J��d�k�k�L�8�9�K�
�7�O�"�"�#�#�%�$�
��&�&�!&���'��E�13�K���(�*�'�!'� ��L�"%�!�$�t�*�|�";�<�!�$�u�+��"<�=�!�$�t�*�|�";�<�!�$�x�.�<�"?�@�	!�
�127��"�"�:�.��{�#�*�*�:�6��"7�����,��K�
�;��&�&�{�3�[<�128%�d�M�e129%�	$�d�M�s
�E!F;�;130G131r�c��SSUSSUS.//S.$)z3Create empty output structure for non-digital PDFs.r�r��nonerr�r�r)r�rls  rr�r��s-���)�$���)E�132����r�__main__rZz1Usage: python pdf_utils.py <pdf_path> [threshold]z9  threshold: line segment merge threshold (default: 0.03)r�zExtracting text from: zLine segment merge threshold: z2--------------------------------------------------)�indent)r)+�__doc__rr��typingrrrrr)r(�ImportError�pypdfrr0�$DEFAULT_LINE_SEGMENT_MERGE_THRESHOLDr�rrrr$�boolr5rOrk�floatrxr�r�r��bytesr�r��__name__�sys�jsonr,�argv�print�exitr%�	thresholdr��dumpsrrr�<module>r�sk��� �	�'�'�!����133���O�(,�$���%��%�1347�T�#�s�(�^�7�D��c��N�7�s�7�@�S��T��>�D��d�3��8�n�!5�6��4��T�#�s�(�^�@T�;U��8<���c�3�h�� �<��<��<�135�$�s�C�x�.��	<�D+/�#���c�3�h�� �#��#�#(�#�136�$�t�C��H�~�137��	#�L ���T�#�s�(�^�,�-� �� �138�$�s�C�x�.�� �P+/���d�3��8�n�%�������	�139#(��140�$�s�C�x�.��
�8+O�\��\�"'�\�141�#�s�(�^�\�D+O�T��T��T�#(�T�142�#�s�(�^�	T�n
�3�
�e�
�PT�UX�Z]�U]�P^�
�(�z����143�3�8�8�}�q��
�A�B�
�I�J�������x�x��{�H�&)�#�(�(�m�a�&7��c�h�h�q�k�"�=a�I�	�"�8�*�144-�.�	�*�9�+�1456�7�	�(�O�
"�8�Y�
7�F�	�$�*�*�V�A�146&�'�#��G�!� ��!�����O��s"�G+�G:�+G7�6G7�:H�H