Ë
    µŒjî  ã                  ó2  — d Z ddlmZ ddlZddlZddlZddlZddlZddlmZ ddl	m
Z
 ddlmZ ddlmZmZmZmZmZmZmZmZmZmZmZ ddlmZ ddlZddlZdd	lmZ dd
lm Z  ddl!m"Z" ddl#m$Z$m%Z% erddl&Z&ddl'Z'ddl(Z(ddl)Z)ddl*m+Z+ g d¢Z,g d¢Z-	 	 	 	 d)d„Z. ej^                  e0«      Z1dZ2dZ3dZ4dZ5h d£Z6d*d„Z7d+d„Z8d+d„Z9ddgZ:d,d„Z; G d„ de «      Z< G d„ de «      Z= G d„ d e «      Z> G d!„ d"e «      Z? G d#„ d$e «      Z@ G d%„ d&e «      ZA G d'„ d(e «      ZBy)-z(Module contains common parsers for PDFs.é    )ÚannotationsN)Údatetime)ÚPath)ÚTemporaryDirectory)ÚTYPE_CHECKINGÚAnyÚBinaryIOÚIterableÚIteratorÚLiteralÚMappingÚOptionalÚSequenceÚUnionÚcast)Úurlparse)ÚDocument)ÚBaseBlobParser)ÚBlob)ÚBaseImageBlobParserÚRapidOCRBlobParser)ÚTextLinearizationConfig)Ú	DCTDecodeÚDCTÚ	JPXDecode)Ú	LZWDecodeÚLZWÚFlateDecodeÚFlÚASCII85DecodeÚA85ÚASCIIHexDecodeÚAHxÚRunLengthDecodeÚRLÚCCITTFaxDecodeÚCCFÚJBIG2Decodec                óÖ   — 	 ddl m}  |«       }d}| D ]6  } ||«      \  }}|sŒ|D �cg c]  }|d   ‘Œ	 }}dj                  |«      z  }Œ8 |S # t        $ r t        d«      ‚w xY wc c}w )zìExtract text from images with RapidOCR.

    Args:
        images: Images to extract text from.

    Returns:
        Text extracted from images.

    Raises:
        ImportError: If `rapidocr-onnxruntime` package is not installed.
    r   )ÚRapidOCRzc`rapidocr-onnxruntime` package not found, please install it with `pip install rapidocr-onnxruntime`Ú é   Ú
)Úrapidocr_onnxruntimer*   ÚImportErrorÚjoin)Úimagesr*   ÚocrÚtextÚimgÚresultÚ_s          úz/var/www/html/Fitness-lenito-AI-main/venv/lib/python3.12/site-packages/langchain_community/document_loaders/parsers/pdf.pyÚ!extract_from_images_with_rapidocrr8   @   s‹   € ð
Ý1ñ ‹*€CØ€DÛˆÙ˜“H‰	ˆ�ÚÙ*0Ó1©& $�d˜1“g¨&ˆFÐ1Ø�D—I‘I˜fÓ%Ñ%‰Dð	 ð
 €Køô ò 
Üð1ó
ð 	
ð
üò 2s   ‚A ¨A&ÁA#z

{image_text}

r-   z
>   ÚsourceÚcreatorÚproducerÚtotal_pagesÚcreationdatec                ó´   — |rU| j                   xs d}|dk(  r|j                  dd«      }d|› d|› d�}|S |dk(  rd	t        j                  |d
¬«      › d|› d�}|S )a®  Format the content of the image with the source of the blob.

    blob: The blob containing the image.
    format::
      The format for the parsed output.
      - "text" = return the content as is
      - "markdown-img" = wrap the content into an image markdown link, w/ link
      pointing to (`![body)(#)`]
      - "html-img" = wrap the content as the `alt` text of an tag and link to
      (`<img alt="{body}" src="#"/>`)
    Ú#zmarkdown-imgÚ]z\\]z![z](Ú)zhtml-imgz
<img alt="T)Úquotez src="z" />)r9   ÚreplaceÚhtmlÚescape)ÚblobÚcontentÚformatr9   s       r7   Ú_format_inner_imagerI   i   sx   € ñ Ø—‘Ò# ˆØ�^Ò#Ø—o‘o c¨6Ó2ˆGØ˜7˜) 2 f X¨QÐ/ˆGð €Nð �zÒ!Ø"¤4§;¡;¨w¸dÔ#CÐ"DÀFÈ6È(ÐRVÐWˆGØ€Nó    c                ó¸   — t         j                  | j                  «       «      st        d«      ‚t	        | j                  dd«      t        «      st        d«      ‚| S )z÷Validate that the metadata has all the standard keys and the page is an integer.

    The standard keys are:
    - source
    - total_page
    - creationdate
    - creator
    - producer

    Validate that page is an integer if it is present.
    z3The PDF parser must valorize the standard metadata.Úpager   z(The PDF metadata page must be a integer.)Ú_STD_METADATA_KEYSÚissubsetÚkeysÚ
ValueErrorÚ
isinstanceÚgetÚint)Úmetadatas    r7   Ú_validate_metadatarU      sJ   € ô ×&Ñ& x§}¡}£Ô7ÜÐNÓOÐOÜ�h—l‘l 6¨1Ó-¬sÔ3ÜÐCÓDÐDØ€OrJ   c                ó  — i }dddœ}| j                  «       D ]×  \  }}t        |«      t        t        fvrt        |«      }|j	                  d«      r|dd }|j                  «       }|dv r:	 t        j                  |j                  dd	«      d
«      j                  d«      ||<   ŒŒ||v r||||   <   |||<   Œžt        |t        «      r|j                  «       ||<   ŒÂt        |t        «      sŒÓ|||<   ŒÙ |S # t        $ r |||<   Y Œìw xY w)zÖPurge metadata from unwanted keys and normalize key names.

    Args:
        metadata: The original metadata dictionary.

    Returns:
        The cleaned and normalized the key format of metadata dictionary.
    r<   r9   )Ú
page_countÚ	file_pathÚ/r,   N)r=   ÚmoddateÚ'r+   zD:%Y%m%d%H%M%S%zÚT)ÚitemsÚtypeÚstrrS   Ú
startswithÚlowerr   ÚstrptimerC   Ú	isoformatrP   rQ   Ústrip)rT   Únew_metadataÚmap_keyÚkÚvs        r7   Ú_purge_metadatari   ’   s  € ð $&€Là#Øñ€Gð —‘Ö ‰ˆˆ1Ü�‹7œ3¤˜*Ñ$Ü�A“ˆAØ�<‰<˜ÔØ�!�"�ˆAØ�G‰G‹IˆØÐ+Ñ+ð$Ü"*×"3Ñ"3Ø—I‘I˜c 2Ó&Ð(:ó#ç‘)˜C“.ð ˜Q’ð
 �'‰\à'(ˆL˜ ™Ñ$ØˆL˜ŠOÜ˜œ3ÔØŸg™g›iˆL˜ŠOÜ˜œ3ÕØˆL˜ŠOð) !ð* Ðøô ò $Ø"#�˜Q“ð$ús   Á+8C4Ã4DÄDz


ú

c                óž   ‡— 	 	 	 	 	 	 	 	 dˆfd„Š ‰| |d«      }|s1d}dj                  t        d„ | «      «      }|rt        d   |z   }||z   }|S )a5  Insert extras such as image/table in a text between two paragraphs if possible,
    else at the end of the text.

    Args:
        extras: List of extra content (images/tables) to insert.
        text_from_page: The text content from the page.

    Returns:
        The merged text with extras inserted.
    c                óþ   •— | rwt         D ]j  }|j                  |«      }|dk7  sŒd }|r ‰	| |d | d«      }|r	|||d  z   }n3d}dj                  t        d„ | «      «      }|r||z   }|d | |z   ||d  z   } |S  d }|S |}|S )NéÿÿÿÿFr+   rj   c                ó   — | S ©N© ©Úxs    r7   Ú<lambda>zO_merge_text_and_extras.<locals>._recurs_merge_text_and_extras.<locals>.<lambda>Û   s   € Á!rJ   )Ú_PARAGRAPH_DELIMITERÚrfindr0   Úfilter)
ÚextrasÚtext_from_pageÚrecursÚdelimÚposÚprevious_textÚall_textÚ
all_extrasÚ
str_extrasÚ_recurs_merge_text_and_extrass
            €r7   r€   z=_merge_text_and_extras.<locals>._recurs_merge_text_and_extrasÊ   sÌ   ø€ ñ ß-�Ø$×*Ñ*¨5Ó1�Ø˜"“9à$(�MÙÙ(EØ" N°4°CÐ$8¸%ó)˜ñ %Ø#0°>À#À$Ð3GÑ#G™à%'˜
Ø%+§[¡[´¹ÀVÓ1LÓ%M˜
Ù%Ø).°Ñ);˜Jà*¨4¨CÐ0°:Ñ=ÀÈsÈtÐ@TÑTð !ð ð
 ˆð1 .ð*  �ð ˆð &ˆHØˆrJ   Tr+   rj   c                ó   — | S ro   rp   rq   s    r7   rs   z(_merge_text_and_extras.<locals>.<lambda>ë   s   € ±!rJ   rm   )rw   ú	list[str]rx   r_   ry   ÚboolÚreturnúOptional[str])r0   rv   rt   )rw   rx   r}   r~   r   r€   s        @r7   Ú_merge_text_and_extrasr†   ¾   sv   ø€ ðØðØ+.ðØ8<ðà	õñ< -¨V°^ÀTÓJ€HÙØˆ
Ø—[‘[¤©°VÓ!<Ó=ˆ
ÙÜ-¨bÑ1°JÑ>ˆJØ! JÑ.ˆà€OrJ   c                  óh   ‡ — e Zd ZdZ	 	 d
dedddddœ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dˆ fd„Zdd„Zdd	„Zˆ xZS )ÚPyPDFParsera  Parse a blob from a PDF using `pypdf` library.

    This class provides methods to parse a blob from a PDF document, supporting various
    configurations such as handling password-protected PDFs, extracting images.
    It integrates the 'pypdf' library for PDF processing and offers synchronous blob
    parsing.

    Examples:
        Setup:

        .. code-block:: bash

            pip install -U langchain-community pypdf

        Load a blob from a PDF file:

        .. code-block:: python

            from langchain_core.documents.base import Blob

            blob = Blob.from_path("./example_data/layout-parser-paper.pdf")

        Instantiate the parser:

        .. code-block:: python

            from langchain_community.document_loaders.parsers import PyPDFParser

            parser = PyPDFParser(
                # password = None,
                mode = "single",
                pages_delimiter = "
",
                # images_parser = TesseractBlobParser(),
            )

        Lazily parse the blob:

        .. code-block:: python

            docs = []
            docs_lazy = parser.lazy_parse(blob)

            for doc in docs_lazy:
                docs.append(doc)
            print(docs[0].page_content[:100])
            print(docs[0].metadata)
    NrL   r3   Úplain)ÚmodeÚpages_delimiterÚimages_parserÚimages_inner_formatÚextraction_modeÚextraction_kwargsc               óÔ   •— t         ‰	| �  «        |dvrt        d«      ‚|| _        |r|s
t	        «       }|| _        || _        || _        || _        || _	        || _
        |xs i | _        y)uý  Initialize a parser based on PyPDF.

        Args:
            password: Optional password for opening encrypted PDFs.
            extract_images: Whether to extract images from the PDF.
            mode: The extraction mode, either "single" for the entire document or "page"
                for page-wise extraction.
            pages_delimiter: A string delimiter to separate pages in single-mode
                extraction.
            images_parser: Optional image blob parser.
            images_inner_format: The format for the parsed output.
                - "text" = return the content as is
                - "markdown-img" = wrap the content into an image markdown link, w/ link
                pointing to (`![body)(#)`]
                - "html-img" = wrap the content as the `alt` text of an tag and link to
                (`<img alt="{body}" src="#"/>`)
            extraction_mode: â€œplainâ€� for legacy functionality, â€œlayoutâ€� extract text
                in a fixed width format that closely adheres to the rendered layout in
                the source pdf.
            extraction_kwargs: Optional additional parameters for the extraction
                process.

        Raises:
            ValueError: If the `mode` is not "single" or "page".
        ©ÚsinglerL   úmode must be single or pageN)ÚsuperÚ__init__rP   Úextract_imagesr   rŒ   r�   ÚpasswordrŠ   r‹   rŽ   r�   )
Úselfr—   r–   rŠ   r‹   rŒ   r�   rŽ   r�   Ú	__class__s
            €r7   r•   zPyPDFParser.__init__$  sv   ø€ ôJ 	‰ÑÔØÐ)Ñ)ÜÐ:Ó;Ð;Ø,ˆÔÙ¡-Ü.Ó0ˆMØ*ˆÔØ#6ˆÔ Ø ˆŒØˆŒ	Ø.ˆÔØ.ˆÔØ!2Ò!8°bˆÕrJ   c              #  ó@  ‡ ‡K  — 	 ddl Šdˆˆ fd„}|j                  «       5 } ‰j                  |‰ j                  ¬«      }t        dddd	œt        t        |j                  xs i «      z  |j                  t        |j                  «      d
œz  «      }g }t        |j                  «      D ]†  \  }} ||¬«      }	‰ j                  |«      }
t        |
g|	«      j                  «       }‰ j                   dk(  r,t#        |t%        |||j&                  |   dœz  «      ¬«      –— Œv|j)                  |«       Œˆ ‰ j                   dk(  r1t#        ‰ j*                  j-                  |«      t%        |«      ¬«      –— ddd«       y# t        $ r t        d«      ‚w xY w# 1 sw Y   yxY w­w)ám  
        Lazily parse the blob.
        Insert image, if possible, between two paragraphs.
        In this way, a paragraph can be continued on the next page.

        Args:
            blob: The blob to parse.

        Raises:
            ImportError: If the `pypdf` package is not found.

        Yield:
            An iterator over the parsed documents.
        r   NzE`pypdf` package not found, please install it with `pip install pypdf`rL   c                óª   •— ‰j                   j                  d«      r| j                  «       S  | j                  dd‰j                  i‰j                  ¤ŽS )zÛ
            Extract text from image given the version of pypdf.

            Args:
                page: The page object to extract text from.

            Returns:
                str: The extracted text.
            Ú3rŽ   rp   )Ú__version__r`   Úextract_textrŽ   r�   )rL   Úpypdfr˜   s    €€r7   Ú_extract_text_from_pagez7PyPDFParser.lazy_parse.<locals>._extract_text_from_pagem  sY   ø€ ð × Ñ ×+Ñ+¨CÔ0Ø×(Ñ(Ó*Ð*à(�t×(Ñ(ñ Ø$(×$8Ñ$8ðà×,Ñ,ñð rJ   ©r—   ÚPyPDFr+   ©r;   r:   r=   )r9   r<   )rL   )rL   Ú
page_label©Úpage_contentrT   r’   )rL   zpypdf.PageObjectr„   r_   )r    r/   Úas_bytes_ioÚ	PdfReaderr—   ri   r   ÚdictrT   r9   ÚlenÚpagesÚ	enumerateÚextract_images_from_pager†   rd   rŠ   r   rU   Úpage_labelsÚappendr‹   r0   )r˜   rF   r¡   Úpdf_file_objÚ
pdf_readerÚdoc_metadataÚsingle_textsÚpage_numberrL   rx   Úimages_from_pager}   r    s   `           @r7   Ú
lazy_parsezPyPDFParser.lazy_parseW  sŸ  ùè ø€ ð	Ûö	ð$ ×ÑÔ <Ø(˜Ÿ™¨ÀÇÁÔNˆJä*Ø$°È"ÑMÜ”t˜Z×0Ñ0Ò6°BÓ7ñ8ð #Ÿk™kÜ#& z×'7Ñ'7Ó#8ññóˆLð ˆLÜ%.¨z×/?Ñ/?Ö%@Ñ!�˜TÙ!8¸dÔ!C�Ø#'×#@Ñ#@ÀÓ#FÐ Ü1Ø%Ð&¨óç‘%“'ð ð —9‘9 Ò&Ü"Ø%-Ü!3Ø(à(3Ø.8×.DÑ.DÀ[Ñ.Qññó"ô	ó 	ð !×'Ñ'¨Õ1ð% &Að& �y‰y˜HÒ$ÜØ!%×!5Ñ!5×!:Ñ!:¸<Ó!HÜ/°Ó=ôò ÷A  Ðøô/ ò 	ÜØWóð ð	ú÷.  Ðüs3   „F†E: ŠF¡EFÅ1	FÅ:FÆFÆFÆFc           	     ó   — | j                   syddl}ddlm} dt	        t
        |d   «      j                  «       vry|d   d   j                  «       }g }|D �]ó  }d}||   d   dk(  sŒt        ||   d	   «      |j                  j                  j                  u r||   d	   d
d n||   d	   d   d
d }|t        v rX||   d   ||   d   }
}	t        j                  ||   j                  «       t        j                   ¬«      j#                  |	|
d«      }nf|t$        v rIt        j&                  |j)                  t+        j,                  ||   j                  «       «      «      «      }nt.        j1                  d«       |€�Œ&t+        j,                  «       }|j3                  «       j4                  dk(  r�ŒY|j7                  |«      j9                  |d¬«       t;        j<                  |j?                  «       d¬«      }tA        | j                   jC                  |«      «      jD                  }|jG                  tI        ||| jJ                  «      «       �Œö tL        jO                  tP        jS                  tU        d|«      «      ¬«      S )úðExtract images from a PDF page and get the text using images_to_text.

        Args:
            page: The page object from which to extract images.

        Returns:
            str: The extracted text from the images on the page.
        r+   r   N©ÚImagez/XObjectz
/Resourcesz/Subtypez/Imagez/Filterr,   z/Heightz/Width©Údtyperm   úUnknown PDF Filter!ÚPNG)rH   z	image/png©Ú	mime_type©Ú
image_text)+rŒ   r    ÚPILr»   r   rª   rO   Ú
get_objectr^   ÚgenericÚ_baseÚ
NameObjectÚ_PDF_FILTER_WITHOUT_LOSSÚnpÚ
frombufferÚget_dataÚuint8ÚreshapeÚ_PDF_FILTER_WITH_LOSSÚarrayÚopenÚioÚBytesIOÚloggerÚwarningÚ	getbufferÚnbytesÚ	fromarrayÚsaver   Ú	from_dataÚgetvalueÚnextr·   r§   r°   rI   r�   Ú_FORMAT_IMAGE_STRrH   Ú_JOIN_IMAGESr0   rv   )r˜   rL   r    r»   ÚxObjectr1   ÚobjÚnp_imageÚ
img_filterÚheightÚwidthÚimage_bytesrF   rÃ   s                 r7   r®   z$PyPDFParser.extract_images_from_page¤  s=  € ð ×!Ò!ØÛÝàœT¤$¨¨\Ñ(:Ó;×@Ñ@ÓBÑBØà�|Ñ$ ZÑ0×;Ñ;Ó=ˆØˆÜˆCØ ˆHØ�s‰|˜JÑ'¨8Ó3ô ˜G C™L¨Ñ3Ó4¸¿¹×8KÑ8K×8VÑ8VÑVð ˜C‘L Ñ+¨A¨BÑ/à  ™ iÑ0°Ñ3°A°BÐ7ð ð
 Ô!9Ñ9Ø$+¨C¡L°Ñ$;¸WÀS¹\È(Ñ=S˜E�Fä!Ÿ}™}Ø ™×-Ñ-Ó/´r·x±xô ç‘g˜f e¨RÓ0ñ ð  Ô#8Ñ8Ü!Ÿx™x¨¯
©
´2·:±:¸gÀc¹l×>SÑ>SÓ>UÓ3VÓ(WÓX‘Hô —N‘NÐ#8Ô9ØÒ'Ü"$§*¡*£,�Kà"×,Ñ,Ó.×5Ñ5¸Ò:Ù à—O‘O HÓ-×2Ñ2°;ÀuÐ2ÔMÜŸ>™>¨+×*>Ñ*>Ó*@ÈKÔX�DÜ!% d×&8Ñ&8×&CÑ&CÀDÓ&IÓ!J×!WÑ!W�JØ—M‘MÜ+¨D°*¸d×>VÑ>VÓWöð9 ô> !×'Ñ'Ü#×(Ñ(¬°°fÓ)=Ó>ð (ó 
ð 	
rJ   ©NF)r—   zOptional[Union[str, bytes]]r–   rƒ   rŠ   úLiteral['single', 'page']r‹   r_   rŒ   úOptional[BaseImageBlobParser]r�   ú+Literal['text', 'markdown-img', 'html-img']rŽ   zLiteral['plain', 'layout']r�   úOptional[dict[str, Any]]©rF   r   r„   úIterator[Document])rL   zpypdf._page.PageObjectr„   r_   )	Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú_DEFAULT_PAGES_DELIMITERr•   r·   r®   Ú__classcell__©r™   s   @r7   rˆ   rˆ   ó   s‹   ø„ ñ.ðd 15Ø$ð19ð
 +1Ø7Ø7;ØKQØ6=Ø6:ñ19à-ð19ð ð19ð
 (ð19ð ð19ð 5ð19ð Ið19ð 4ð19ð 4õ19ófK÷Z4
rJ   rˆ   c                  óž   ‡ — e Zd ZdZdZ	 dddeddddœ	 	 	 	 	 	 	 	 	 	 	 	 	 dˆ fd„Zedd„«       Zedd	„«       Z		 	 d	 	 	 	 	 	 	 dd
„Z
dd„Zˆ xZS )ÚPDFMinerParsera…  Parse a blob from a PDF using `pdfminer.six` library.

    This class provides methods to parse a blob from a PDF document, supporting various
    configurations such as handling password-protected PDFs, extracting images, and
    defining extraction mode.
    It integrates the 'pdfminer.six' library for PDF processing and offers synchronous
    blob parsing.

    Examples:
        Setup:

        .. code-block:: bash

            pip install -U langchain-community pdfminer.six pillow

        Load a blob from a PDF file:

        .. code-block:: python

            from langchain_core.documents.base import Blob

            blob = Blob.from_path("./example_data/layout-parser-paper.pdf")

        Instantiate the parser:

        .. code-block:: python

            from langchain_community.document_loaders.parsers import PDFMinerParser

            parser = PDFMinerParser(
                # password = None,
                mode = "single",
                pages_delimiter = "
",
                # extract_images = True,
                # images_to_text = convert_images_to_text_with_tesseract(),
            )

        Lazily parse the blob:

        .. code-block:: python

            docs = []
            docs_lazy = parser.lazy_parse(blob)

            for doc in docs_lazy:
                docs.append(doc)
            print(docs[0].page_content[:100])
            print(docs[0].metadata)
    FNr’   r3   )r—   rŠ   r‹   rŒ   r�   Úconcatenate_pagesc               ó,  •— t         ‰| �  «        |dvrt        d«      ‚|r|s
t        «       }|| _        || _        || _        || _        || _        || _	        |�<t        j                  s dt        _        t        j                  d«       |rdnd| _        yy)aH  Initialize a parser based on PDFMiner.

        Args:
            password: Optional password for opening encrypted PDFs.
            mode: Extraction mode to use. Either "single" or "page" for page-wise
                extraction.
            pages_delimiter: A string delimiter to separate pages in single-mode
                extraction.
            extract_images: Whether to extract images from PDF.
            images_inner_format: The format for the parsed output.
                - "text" = return the content as is
                - "markdown-img" = wrap the content into an image markdown link, w/ link
                pointing to (`![body)(#)`]
                - "html-img" = wrap the content as the `alt` text of an tag and link to
                (`<img alt="{body}" src="#"/>`)
            concatenate_pages: Deprecated. If True, concatenate all PDF pages
                into one a single document. Otherwise, return one document per page.

        Returns:
            This method does not directly return data. Use the `parse` or `lazy_parse`
            methods to retrieve parsed documents with content and metadata.

        Raises:
            ValueError: If the `mode` is not "single" or "page".

        Warnings:
            `concatenate_pages` parameter is deprecated. Use `mode='single' or 'page'
            instead.
        r‘   r“   NTzS`concatenate_pages` parameter is deprecated. Use `mode='single' or 'page'` instead.r’   rL   )r”   r•   rP   r   r–   rŒ   r�   r—   rŠ   r‹   rõ   Ú_warn_concatenate_pagesrÔ   rÕ   )	r˜   r–   r—   rŠ   r‹   rŒ   r�   rö   r™   s	           €r7   r•   zPDFMinerParser.__init__  s›   ø€ ôP 	‰ÑÔØÐ)Ñ)ÜÐ:Ó;Ð;Ù¡-Ü.Ó0ˆMØ,ˆÔØ*ˆÔØ#6ˆÔ Ø ˆŒØˆŒ	Ø.ˆÔØÐ(Ü!×9Ò9Ø9=”Ô6Ü—‘ð=ôñ %6™¸6ˆD�Ið )rJ   c                óî   ‡— ddl mŠ t        | t        «      r!| j	                  d«      rt        | dd dd«      S 	 d„ | D «       }d	j                  ˆfd
„|D «       «      S # t        $ r t        | «      cY S w xY w)zæ
        Decodes a PDFDocEncoding string to Unicode.
        Adds py3 compatibility to pdfminer's version.

        Args:
            s: The string to decode.

        Returns:
            str: The decoded Unicode string.
        r   )ÚPDFDocEncodings   þÿé   Nzutf-16beÚignorec              3  óV   K  — | ]!  }t        |t        «      rt        |«      n|–— Œ# y ­wro   )rQ   r_   Úord)Ú.0Úcs     r7   Ú	<genexpr>z-PDFMinerParser.decode_text.<locals>.<genexpr>]  s#   è ø€ ÐCÁ¸Aœj¨¬CÔ0”C˜”F°aÓ7Áùs   ‚')r+   c              3  ó(   •K  — | ]	  }‰|   –— Œ y ­wro   rp   )rÿ   Úorú   s     €r7   r  z-PDFMinerParser.decode_text.<locals>.<genexpr>^  s   øè ø€ Ð;±d°˜>¨!Õ,±dùs   ƒ)Úpdfminer.utilsrú   rQ   Úbytesr`   r_   r0   Ú
IndexError)ÚsÚordsrú   s     @r7   Údecode_textzPDFMinerParser.decode_textL  sn   ø€ õ 	2ä�aœÔ A§L¡L°Ô$=Ü�q˜˜�u˜j¨(Ó3Ð3ð	ÙCÁÓCˆDØ—7‘7Ó;±dÓ;Ó;Ð;øÜò 	Ü�q“6ŠMð	ús   º"A ÁA4Á3A4c                óà  — ddl m} t        | d«      r| j                  «       } t	        | t
        «      r#t        t        t        j                  | «      «      S t	        | |«      rt        j                  | j                  «      S t	        | t        t        f«      rt        j                  | «      S t	        | t        «      r2| j                  «       D ]  \  }}t        j                  |«      | |<   Œ | S | S )zÒ
        Recursively resolve the metadata values.

        Args:
            obj: The object to resolve and decode. It can be of any type.

        Returns:
            The resolved and decoded object.
        r   )Ú	PSLiteralÚresolve)Úpdfminer.psparserr  Úhasattrr  rQ   ÚlistÚmaprõ   Úresolve_and_decoder	  Únamer_   r  rª   r]   )rà   r  rg   rh   s       r7   r  z!PDFMinerParser.resolve_and_decodeb  s¸   € õ 	0ä�3˜	Ô"Ø—+‘+“-ˆCÜ�cœ4Ô ÜœœN×=Ñ=¸sÓCÓDÐDÜ˜˜YÔ'Ü!×-Ñ-¨c¯h©hÓ7Ð7Ü˜œc¤5˜\Ô*Ü!×-Ñ-¨cÓ2Ð2Ü˜œTÔ"ØŸ	™	ž‘��1Ü'×:Ñ:¸1Ó=��A’ð $àˆJàˆ
rJ   c           	     ó¢  — ddl m}m}m}  ||«      } ||||¬«      }i }	|j                  D ]  }
|	j                  |
«       Œ |	j                  «       D ]  \  }}	 t        j                  |«      |	|<   Œ  t        t        |j                  |«      «      «      |	d<   |	S # t        $ r*}t        j                  d|t        |«      «       Y d}~Œwd}~ww xY w)ag  
        Extract metadata from a PDF file.

        Args:
            fp: The file pointer to the PDF file.
            password: The password for the PDF file, if encrypted. Defaults to an empty
                string.
            caching: Whether to cache the PDF structure. Defaults to True.

        Returns:
            Metadata of the PDF file.
        r   )ÚPDFDocumentÚPDFPageÚ	PDFParser)r—   ÚcachingzD[WARNING] Metadata key "%s" could not be parsed due to exception: %sNr<   )Úpdfminer.pdfpager  r  r  ÚinfoÚupdater]   rõ   r  Ú	ExceptionrÔ   rÕ   r_   r«   r  Úcreate_pages)r˜   Úfpr—   r  r  r  r  ÚparserÚdocrT   r  rg   rh   Úes                 r7   Ú_get_metadatazPDFMinerParser._get_metadata~  sË   € ÷$ 	EÑDñ ˜2“ˆá˜&¨8¸WÔEˆØˆà—H”HˆDØ�O‰O˜DÕ!ð à—N‘NÖ$‰DˆAˆqð
Ü,×?Ñ?ÀÓB�˜’ð %ô #&¤d¨7×+?Ñ+?ÀÓ+DÓ&EÓ"Fˆ�Ñàˆøô ò ô —‘ð$àÜ˜“F÷	ñ ûðús   ÁBÂ	CÂ$ C	Ã	Cc              #  óú  ‡ ‡‡‡‡‡‡K  — 	 ddl }ddlm} ddlm}mŠmŠm}m}m	Šm
Š ddlm}m} ddlm}	 t!        |j"                  «      dk  rt%        d«      ‚	 |j'                  «       5 }
t)        «       5 Š|	j+                  |
‰ j,                  xs d
¬«      } |«       }t/        ddd
dœ‰ j1                  |
‰ j,                  xs d
¬«      z  «      }|j2                  |d<    G ˆˆˆˆˆ ˆˆfd„d|«      }t5        j6                  «       Š || || |«       ¬«      «      }g }t9        |«      D ]Î  \  }}‰j;                  d«       ‰j=                  d«       |j?                  |«       ‰jA                  «       }|jC                  «       }‰ jD                  dk(  r@‰j;                  d«       ‰j=                  d«       tG        |tI        |d|iz  «      ¬«      –— Œ¨|jK                  d«      r|dd }|jM                  |«       ŒÐ ‰ jD                  dk(  r3‰ jN                  jQ                  |«      }tG        |tI        |«      ¬«      –— ddd«       ddd«       y# t$        $ r t%        d	«      ‚w xY w# 1 sw Y   Œ*xY w# 1 sw Y   yxY w­w)a€  
        Lazily parse the blob.
        Insert image, if possible, between two paragraphs.
        In this way, a paragraph can be continued on the next page.

        Args:
            blob: The blob to parse.

        Raises:
            ImportError: If the `pdfminer.six` or `pillow` package is not found.

        Yield:
            An iterator over the parsed documents.
        r   N)ÚPDFLayoutAnalyzer)ÚLAParamsÚLTContainerÚLTImageÚLTItemÚLTPageÚLTTextÚ	LTTextBox)ÚPDFPageInterpreterÚPDFResourceManager)r  i:>4z§This parser is tested with pdfminer.six version 20201018 or later. Remove pdfminer, and install pdfminer.six with `pip uninstall pdfminer && pip install pdfminer.six`.zMpdfminer package not found, please install it with `pip install pdfminer.six`r+   r¢   ÚPDFMinerr¤   r9   c                  óN   •‡ — e Zd Z	 	 d	 	 	 	 	 	 	 dˆ fd„Zdˆˆˆˆˆˆˆfd„Zˆ xZS )ú*PDFMinerParser.lazy_parse.<locals>.Visitorc                ó*   •— t         ‰| �  |||¬«       y )N)ÚpagenoÚlaparams)r”   r•   )r˜   Úrsrcmgrr1  r2  r™   s       €r7   r•   z3PDFMinerParser.lazy_parse.<locals>.Visitor.__init__à  s   ø€ ô ‘GÑ$ W°VÀhÐ$ÕOrJ   c           	     ó2   •‡— dˆˆˆˆˆˆˆˆ	fd„Š ‰|«       y )Nc                óJ  •— t        | ‰«      r| D ]
  } ‰|«       Œ n+t        | ‰	«      r‰j                  | j                  «       «       t        | ‰
«      r‰j                  d«       y t        | ‰«      r±‰j                  r¤ddlm}  |‰«      }|j                  | «      }t        j                  t        ‰«      |z  «      }d|j                  d<   t        ‰j                  j                  |«      «      j                  }‰j                  t        ||‰j                  «      «       y y y )Nr-   r   )ÚImageWriterr?   r9   )rQ   ÚwriteÚget_textrŒ   Úpdfminer.imager6  Úexport_imager   Ú	from_pathr   rT   rÜ   r·   r§   rI   r�   )ÚitemÚchildr6  Úimage_writerÚfilenamerF   rÃ   r%  r&  r)  r*  Úrenderr˜   ÚtempdirÚtext_ios          €€€€€€€€r7   r@  zIPDFMinerParser.lazy_parse.<locals>.Visitor.receive_layout.<locals>.renderé  sø   ø€ Ü% d¨KÔ8Û)- Ù & u¥ñ *.ä'¨¨fÔ5Ø#ŸM™M¨$¯-©-«/Ô:Ü% d¨IÔ6Ø#ŸM™M¨$Õ/Ü'¨¨gÔ6Ø#×1Ò1Ý Fá/:¸7Ó/C Ø+7×+DÑ+DÀTÓ+J Ü'+§~¡~´d¸7³mÀhÑ6NÓ'O Ø:= §¡¨hÑ 7Ü-1Ø$(×$6Ñ$6×$AÑ$AÀ$Ó$Gó."ç".¡,ð !+ð !(§¡Ü$7Ø(,¨j¸$×:RÑ:Ró%&õ!"ð  2ð" !rJ   )r<  r'  r„   ÚNonerp   )
ÚmeÚltpager@  r%  r&  r)  r*  r˜   rA  rB  s
     @€€€€€€€r7   Úreceive_layoutz9PDFMinerParser.lazy_parse.<locals>.Visitor.receive_layoutè  s   ù€ ÷!ô !ñ8 ˜6•NrJ   )r,   N)r3  r,  r1  rS   r2  zOptional[LAParams]r„   rC  )rE  r(  r„   rC  )rí   rî   rï   r•   rF  rò   )r™   r%  r&  r)  r*  r˜   rA  rB  s   @€€€€€€€r7   ÚVisitorr/  ß  sE   ù„ ð #$Ø37ð	Pà/ðPð  ðPð 1ð	Pð
 õP÷#÷ #rJ   rG  )r2  rL   r¦   Úrm   r’   ))ÚpdfminerÚpdfminer.converterr#  Úpdfminer.layoutr$  r%  r&  r'  r(  r)  r*  Úpdfminer.pdfinterpr+  r,  r  r  rS   rž   r/   r¨   r   Ú	get_pagesr—   ri   r!  r9   rÒ   ÚStringIOr­   ÚtruncateÚseekÚprocess_pagerÛ   rd   rŠ   r   rU   Úendswithr°   r‹   r0   )r˜   rF   rI  r#  r$  r'  r(  r+  r,  r  r±   r¬   r3  r³   rG  Úvisitor_for_allÚall_contentÚirL   r}   Údocument_contentr%  r&  r)  r*  rA  rB  s   `                    @@@@@@r7   r·   zPDFMinerParser.lazy_parse¬  sY  þè ø€ ð	ÛÝ<÷÷ ñ ÷ RÝ0ä�8×'Ñ'Ó(¨8Ò3Ü!ðLóð ð 4ð ×ÑÔ <Ô1CÔ1EÈØ×%Ñ% l¸T¿]¹]Ò=PÈbÐ%ÓQˆEÙ(Ó*ˆGÜ*Ø'°JÐPRÑSØ×$Ñ$ \¸D¿M¹MÒ<OÈRÐ$ÓPñQóˆLð &*§[¡[ˆL˜Ñ"÷&#ô &#Ð+ô &#ôP —k‘k“mˆGÙ0Ø™ ±8³:Ô>óˆOð ˆKÜ$ UÖ+‘��4Ø× Ñ  Ô#Ø—‘˜Q”Ø×,Ñ,¨TÔ2à"×+Ñ+Ó-�à#Ÿ>™>Ó+�Ø—9‘9 Ò&Ø×$Ñ$ QÔ'Ø—L‘L ”OÜ"Ø%-Ü!3°LÀFÈAÀ;Ñ4NÓ!Oôó ð
  ×(Ñ(¨Ô.Ø#+¨C¨R =˜Ø×&Ñ& xÕ0ð% ,ð& �y‰y˜HÒ$à#'×#7Ñ#7×#<Ñ#<¸[Ó#IÐ ÜØ!1Ü/°Ó=ôò ÷Y 2F×Ðøô ò 	Üð2óð ð	ú÷ 2FÐ1Eú×ÐüsN   ‰I;‹AI ÁI;Á)I/Á4GI#È:I/É	I;ÉI É I;É#I,	É(I/É/I8É4I;©F)r–   rƒ   r—   r…   rŠ   rç   r‹   r_   rŒ   rè   r�   ré   rö   zOptional[bool])r  zUnion[bytes, str]r„   r_   )rà   r   r„   r   )r+   T)r  r	   r—   r_   r  rƒ   r„   údict[str, Any]rë   )rí   rî   rï   rð   rø   rñ   r•   Ústaticmethodr	  r  r!  r·   rò   ró   s   @r7   rõ   rõ   Û  så   ø„ ñ0ðd $Ðð  %ð:Bð #'Ø*2Ø7Ø7;ØKQØ,0ñ:Bàð:Bð  ð	:Bð
 (ð:Bð ð:Bð 5ð:Bð Ið:Bð *õ:Bðx òó ðð* òó ðð< Øð	,àð,ð ð,ð ð	,ð
 
ó,÷\yrJ   rõ   c            	      óÞ   ‡ — e Zd ZdZ ej
                  «       Z	 	 dddedddddœ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dˆ fd„Zdd„Z		 d	 	 	 	 	 dd„Z
	 	 	 	 	 	 	 	 dd	„Zdd
„Z	 	 	 	 	 	 dd„Zdd„Zˆ xZS )ÚPyMuPDFParsera²  Parse a blob from a PDF using `PyMuPDF` library.

    This class provides methods to parse a blob from a PDF document, supporting various
    configurations such as handling password-protected PDFs, extracting images, and
    defining extraction mode.
    It integrates the 'PyMuPDF' library for PDF processing and offers synchronous blob
    parsing.

    Examples:
        Setup:

        .. code-block:: bash

            pip install -U langchain-community pymupdf

        Load a blob from a PDF file:

        .. code-block:: python

            from langchain_core.documents.base import Blob

            blob = Blob.from_path("./example_data/layout-parser-paper.pdf")

        Instantiate the parser:

        .. code-block:: python

            from langchain_community.document_loaders.parsers import PyMuPDFParser

            parser = PyMuPDFParser(
                # password = None,
                mode = "single",
                pages_delimiter = "
",
                # images_parser = TesseractBlobParser(),
                # extract_tables="markdown",
                # extract_tables_settings=None,
                # text_kwargs=None,
            )

        Lazily parse the blob:

        .. code-block:: python

            docs = []
            docs_lazy = parser.lazy_parse(blob)

            for doc in docs_lazy:
                docs.append(doc)
            print(docs[0].page_content[:100])
            print(docs[0].metadata)
    NrL   r3   )r—   rŠ   r‹   rŒ   r�   Úextract_tablesÚextract_tables_settingsc               ó  •— t         ‰
| �  «        |dvrt        d«      ‚|r|dvrt        d«      ‚|| _        || _        || _        |xs i | _        |r|s
t        «       }|| _        || _	        || _
        || _        |	| _        y)aÓ  Initialize a parser based on PyMuPDF.

        Args:
            password: Optional password for opening encrypted PDFs.
            mode: The extraction mode, either "single" for the entire document or "page"
                for page-wise extraction.
            pages_delimiter: A string delimiter to separate pages in single-mode
                extraction.
            extract_images: Whether to extract images from the PDF.
            images_parser: Optional image blob parser.
            images_inner_format: The format for the parsed output.
                - "text" = return the content as is
                - "markdown-img" = wrap the content into an image markdown link, w/ link
                pointing to (`![body)(#)`]
                - "html-img" = wrap the content as the `alt` text of an tag and link to
                (`<img alt="{body}" src="#"/>`)
            extract_tables: Whether to extract tables in a specific format, such as
                "csv", "markdown", or "html".
            extract_tables_settings: Optional dictionary of settings for customizing
                table extraction.

        Returns:
            This method does not directly return data. Use the `parse` or `lazy_parse`
            methods to retrieve parsed documents with content and metadata.

        Raises:
            ValueError: If the mode is not "single" or "page".
            ValueError: If the extract_tables format is not "markdown", "html",
            or "csv".
        r‘   r“   )ÚmarkdownrD   Úcsvzmode must be markdownN)r”   r•   rP   rŠ   r‹   r—   Útext_kwargsr   r–   r�   rŒ   r\  r]  )r˜   ra  r–   r—   rŠ   r‹   rŒ   r�   r\  r]  r™   s             €r7   r•   zPyMuPDFParser.__init__a  s•   ø€ ôV 	‰ÑÔØÐ)Ñ)ÜÐ:Ó;Ð;Ù˜nÐ4OÑOÜÐ4Ó5Ð5àˆŒ	Ø.ˆÔØ ˆŒØ&Ò,¨"ˆÔÙ¡-Ü.Ó0ˆMØ,ˆÔØ#6ˆÔ Ø*ˆÔØ,ˆÔØ'>ˆÕ$rJ   c                ó$   — | j                  |«      S ro   )Ú_lazy_parse)r˜   rF   s     r7   r·   zPyMuPDFParser.lazy_parsež  s   € Ø×ÑØó
ð 	
rJ   c              #  óø  K  — 	 ddl }|xs | j                  }| j                  sNddlm}m}m}m} i dd“dd“dd“dd“d	d“d
|“dd“dd“d|“dd“dd“dd“d|“d|“dd“dd“dd“ddddddœ¥| _        t        j                  5  |j                  «       5 }|j                  € |j                  |«      }	n |j                  |d¬«      }	|	j                  r|	j                  | j                   «       ddddœ| j#                  |	|«      z  }
g }|	D ]k  }| j%                  |	||«      j'                  «       }| j(                  dk(  r(t+        |t-        |
d|j.                  iz  «      ¬«      –— Œ[|j1                  |«       Œm | j(                  d k(  r1t+        | j2                  j5                  |«      t-        |
«      ¬«      –— ddd«       ddd«       y# t        $ r t        d«      ‚w xY w# 1 sw Y   Œ*xY w# 1 sw Y   yxY w­w)!a  Lazily parse the blob.
        Insert image, if possible, between two paragraphs.
        In this way, a paragraph can be continued on the next page.

        Args:
            blob: The blob to parse.
            text_kwargs: Optional keyword arguments to pass to the `get_text` method.
                If provided at run time, it will override the default text_kwargs.

        Raises:
            ImportError: If the `pypdf` package is not found.

        Yield:
            An iterator over the parsed documents.
        r   N)ÚDEFAULT_JOIN_TOLERANCEÚDEFAULT_MIN_WORDS_HORIZONTALÚDEFAULT_MIN_WORDS_VERTICALÚDEFAULT_SNAP_TOLERANCEÚclipÚvertical_strategyÚlinesÚhorizontal_strategyÚvertical_linesÚhorizontal_linesÚsnap_toleranceÚsnap_x_toleranceÚsnap_y_toleranceÚjoin_toleranceÚjoin_x_toleranceÚjoin_y_toleranceÚedge_min_lengthé   Úmin_words_verticalÚmin_words_horizontalÚintersection_toleranceÚintersection_x_toleranceÚintersection_y_tolerance)Útext_toleranceÚtext_x_toleranceÚtext_y_toleranceÚstrategyÚ	add_lineszGpymupdf package not found, please install it with `pip install pymupdf`Úpdf)ÚstreamÚfiletypeÚPyMuPDFr+   r¤   rL   r¦   r’   )Úpymupdfra  r]  Úpymupdf.tablere  rf  rg  rh  r/   r[  Ú_lockr¨   ÚdatarÑ   Úis_encryptedÚauthenticater—   Ú_extract_metadataÚ_get_page_contentrd   rŠ   r   rU   Únumberr°   r‹   r0   )r˜   rF   ra  r…  re  rf  rg  rh  rX   r  r³   Úfull_contentrL   r}   s                 r7   rc  zPyMuPDFParser._lazy_parse£  s‘  è ø€ ð,)	Ûà%Ò9¨×)9Ñ)9ˆKØ×/Ò/÷ó ð0à˜Dð0ð (¨ð0ð *¨7ð	0ð
 % dð0ð '¨ð0ð %Ð&<ð0ð '¨ð0ð '¨ð0ð %Ð&<ð0ð '¨ð0ð '¨ð0ð & qð0ð )Ð*Dð0ð +Ð,Hð0ð  -¨að!0ð" /°ð#0ð$ /°ð%0ð& '(Ø()Ø()Ø $Ø!%ò/0�Ô,ô> × Ó Ø×!Ñ!Ô# yØ—9‘9Ð$Ø&˜'Ÿ,™, yÓ1‘Cà&˜'Ÿ,™,¨iÀ%ÔH�CØ×#Ò#Ø×$Ñ$ T§]¡]Ô3à )Ø(Ø$&ñ ð ×*Ñ*¨3°Ó5ñ	 6�ð
  "�Û�DØ#×5Ñ5°c¸4ÀÓM×SÑSÓU�HØ—y‘y FÒ*Ü&Ø)1Ü%7Ø ,°¸¿¹Ð/DÑ Dó&ôó ð %×+Ñ+¨HÕ5ð  ð —9‘9 Ò(Ü"Ø%)×%9Ñ%9×%>Ñ%>¸|Ó%LÜ!3°LÓ!Aôò ÷5 $÷ !Ð øô ò 	Üð-óð ð	ú÷ $Ð#ú÷ !Ð üsN   ‚G:„A.G
 Á2G:ÂG.ÂD&G"Æ9G.Ç	G:Ç
GÇG:Ç"G+	Ç'G.Ç.G7Ç3G:c                óô   —  |j                   di i | j                  ¥|¥¤Ž}| j                  ||«      }| j                  |«      }g }|r|j	                  |«       |r|j	                  |«       t        ||«      }|S )a:  Get the text of the page using PyMuPDF and RapidOCR and issue a warning
        if it is empty.

        Args:
            doc: The PyMuPDF document object.
            page: The PyMuPDF page object.
            blob: The blob being parsed.

        Returns:
            str: The text content of the page.
        rp   )r8  ra  Ú_extract_images_from_pageÚ_extract_tables_from_pager°   r†   )	r˜   r  rL   ra  rx   r¶   Útables_from_pagerw   r}   s	            r7   rŒ  zPyMuPDFParser._get_page_content  s‚   € ð" '˜Ÿ™ÑMÐ)L¨D×,<Ñ,<Ð)LÀÐ)LÑMˆØ×9Ñ9¸#¸tÓDÐØ×9Ñ9¸$Ó?ÐØˆÙØ�M‰MÐ*Ô+ÙØ�M‰MÐ*Ô+Ü)¨&°.ÓAˆàˆrJ   c                óX  — t        i ddd|j                  |j                  t        |«      dœ¥|j                  D �ci c]5  }t	        |j                  |   t
        t        f«      r||j                  |   “Œ7 c}¥«      }dD ]#  }||j                  v sŒ|j                  |   ||<   Œ% |S c c}w )z×Extract metadata from the document and page.

        Args:
            doc: The PyMuPDF document object.
            blob: The blob being parsed.

        Returns:
            dict: The extracted metadata.
        r„  r+   )r;   r:   r=   r9   rX   r<   )ÚmodDateÚcreationDate)ri   r9   r«   rT   rQ   r_   rS   )r˜   r  rF   rg   rT   s        r7   r‹  zPyMuPDFParser._extract_metadata!  s¹   € ô #ðà )Ø(Ø$&Ø"Ÿk™kØ!%§¡Ü#& s£8ñðð !Ÿ\š\óá)˜Ü! #§,¡,¨q¡/´C¼°:Ô>ð �s—|‘| A‘Ñ&Ø)ñðó
ˆó" -ˆAØ�C—L‘LÒ Ø!Ÿl™l¨1™o�˜’ð -ð ˆùòs   »:B'
c                ó4  — | j                   syddl}|j                  «       }g }|D �]=  }| j                   sŒ|d   } |j                  ||«      }t	        j
                  |j                  t        j                  ¬«      j                  |j                  |j                  d«      }	t        j                  «       }
|
j                  «       j                  dk(  rŒ¯t        j                   |
|	«       t#        j$                  |
j'                  «       d¬«      }t)        | j                   j+                  |«      «      j,                  }|j/                  t1        ||| j2                  «      «       �Œ@ t4        j7                  t8        j;                  t=        d|«      «      ¬«      S )	a	  Extract images from a PDF page and get the text using images_to_text.

        Args:
            doc: The PyMuPDF document object.
            page: The PyMuPDF page object.

        Returns:
            str: The extracted text from the images on the page.
        r+   r   Nr¼   rm   úapplication/x-npyrÀ   rÂ   )rŒ   r…  Ú
get_imagesÚPixmaprÊ   rË   ÚsamplesrÍ   rÎ   rã   rä   rÒ   rÓ   rÖ   r×   ÚnumpyrÙ   r   rÚ   rÛ   rÜ   r·   r§   r°   rI   r�   rÝ   rH   rÞ   r0   rv   )r˜   r  rL   r…  Úimg_listr1   r4   ÚxrefÚpixÚimagerå   rF   rÃ   s                r7   r�  z'PyMuPDFParser._extract_images_from_pageA  sF  € ð ×!Ò!ØÛà—?‘?Ó$ˆØˆÜˆCØ×!Ó!Ø˜1‘v�Ø$�g—n‘n S¨$Ó/�ÜŸ™ c§k¡k¼¿¹ÔB×JÑJØ—J‘J §	¡	¨2ó�ô !Ÿj™j›l�Ø×(Ñ(Ó*×1Ñ1°QÒ6Øä—
‘
˜;¨Ô.Ü—~‘~Ø×(Ñ(Ó*Ð6Iô�ô " $×"4Ñ"4×"?Ñ"?ÀÓ"EÓF×SÑS�
à—‘Ü'¨¨j¸$×:RÑ:RÓSöð# ô( !×'Ñ'Ü#×(Ñ(¬°°fÓ)=Ó>ð (ó 
ð 	
rJ   c           
     ó   — | j                   €yddl}t         |j                  j                  |fi | j
                  ¤Ž«      }|rü| j                   dk(  r1t        j                  |D �cg c]  }|j                  «       ‘Œ c}«      S | j                   dk(  rCt        j                  |D �cg c]$  }|j                  «       j                  ddd¬«      ‘Œ& c}«      S | j                   dk(  rBt        j                  |D �cg c]#  }|j                  «       j                  dd¬	«      ‘Œ% c}«      S t        d
| j                   › d�«      ‚yc c}w c c}w c c}w )z³Extract tables from a PDF page.

        Args:
            page: The PyMuPDF page object.

        Returns:
            str: The extracted tables in the specified format.
        Nr+   r   r_  rD   F)ÚheaderÚindexÚ	bold_rowsr`  )r¡  r¢  zextract_tables z not implemented)r\  r…  r  ÚtableÚfind_tablesr]  Ú_JOIN_TABLESr0   Úto_markdownÚ	to_pandasÚto_htmlÚto_csvrP   )r˜   rL   r…  Útables_listr¤  s        r7   r‘  z'PyMuPDFParser._extract_tables_from_pagek  sv  € ð ×ÑÐ&ØÛäØ%ˆG�M‰M×%Ñ% dÑK¨d×.JÑ.JÑKó
ˆñ Ø×"Ñ" jÒ0Ü#×(Ñ(É;Ó)WÉ;À%¨%×*;Ñ*;Õ*=È;Ñ)WÓXÐXØ×$Ñ$¨Ò.Ü#×(Ñ(ñ &1óñ &1˜Eð Ÿ™Ó)×1Ñ1Ø#(Ø"'Ø&+ð 2õ ð
 &1ñó	ð 	ð ×$Ñ$¨Ò-Ü#×(Ñ(ñ &1óñ
 &1˜Eð	 Ÿ™Ó)×0Ñ0Ø#(Ø"'ð 1õ ð &1ñóð ô !Ø% d×&9Ñ&9Ð%:Ð:JÐKóð ð ùò5 *Xùòùòs   Á&EÂ&)EÃ8(Eræ   )ra  rê   r–   rƒ   r—   r…   rŠ   rç   r‹   r_   rŒ   rè   r�   ré   r\  z/Union[Literal['csv', 'markdown', 'html'], None]r]  rê   r„   rC  rë   ro   )rF   r   ra  rê   r„   rì   )r  úpymupdf.DocumentrL   úpymupdf.Pagera  rX  r„   r_   )r  r¬  rF   r   r„   rª   )r  r¬  rL   r­  r„   r_   )rL   r­  r„   r_   )rí   rî   rï   rð   Ú	threadingÚLockr‡  rñ   r•   r·   rc  rŒ  r‹  r�  r‘  rò   ró   s   @r7   r[  r[  (  s+  ø„ ñ2ðl ˆI�N‰NÓ€Eð 15Ø$ð;?ð
 #'Ø*0Ø7Ø7;ØKQØJNØ<@ñ;?à-ð;?ð ð;?ð
  ð;?ð (ð;?ð ð;?ð 5ð;?ð Ið;?ð Hð;?ð ":ð;?ð 
õ;?óz
ð 15ð_àð_ð
 .ð_ð 
ó_ðBàðð ðð $ð	ð
 
óó:ð@(
Ø#ð(
Ø+7ð(
à	ó(
÷T,rJ   r[  c                  ó‚   ‡ — e Zd ZdZ ej
                  «       Z	 d	ddedddœ	 	 	 	 	 	 	 	 	 	 	 	 	 d
ˆ fd„Zdd„Z	dd„Z
ˆ xZS )ÚPyPDFium2Parserao  Parse a blob from a PDF using `PyPDFium2` library.

    This class provides methods to parse a blob from a PDF document, supporting various
    configurations such as handling password-protected PDFs, extracting images, and
    defining extraction mode.
    It integrates the 'PyPDFium2' library for PDF processing and offers synchronous
    blob parsing.

    Examples:
        Setup:

        .. code-block:: bash

            pip install -U langchain-community pypdfium2

        Load a blob from a PDF file:

        .. code-block:: python

            from langchain_core.documents.base import Blob

            blob = Blob.from_path("./example_data/layout-parser-paper.pdf")

        Instantiate the parser:

        .. code-block:: python

            from langchain_community.document_loaders.parsers import PyPDFium2Parser

            parser = PyPDFium2Parser(
                # password=None,
                mode="page",
                pages_delimiter="
",
                # extract_images = True,
                # images_to_text = convert_images_to_text_with_tesseract(),
            )

        Lazily parse the blob:

        .. code-block:: python

            docs = []
            docs_lazy = parser.lazy_parse(blob)

            for doc in docs_lazy:
                docs.append(doc)
            print(docs[0].page_content[:100])
            print(docs[0].metadata)
    NrL   r3   )r—   rŠ   r‹   rŒ   r�   c               ó°   •— t         ‰| �  «        |dvrt        d«      ‚|| _        |r|s
t	        «       }|| _        || _        || _        || _        || _	        y)uk  Initialize a parser based on PyPDFium2.

        Args:
            password: Optional password for opening encrypted PDFs.
            mode: The extraction mode, either "single" for the entire document or "page"
                for page-wise extraction.
            pages_delimiter: A string delimiter to separate pages in single-mode
                extraction.
            extract_images: Whether to extract images from the PDF.
            images_parser: Optional image blob parser.
            images_inner_format: The format for the parsed output.
                - "text" = return the content as is
                - "markdown-img" = wrap the content into an image markdown link, w/ link
                pointing to (`![body)(#)`]
                - "html-img" = wrap the content as the `alt` text of an tag and link to
                (`<img alt="{body}" src="#"/>`)
            extraction_mode: â€œplainâ€� for legacy functionality, â€œlayoutâ€� for experimental
                layout mode functionality
            extraction_kwargs: Optional additional parameters for the extraction
                process.

        Returns:
            This method does not directly return data. Use the `parse` or `lazy_parse`
            methods to retrieve parsed documents with content and metadata.

        Raises:
            ValueError: If the mode is not "single" or "page".
        r‘   r“   N)
r”   r•   rP   r–   r   rŒ   r�   r—   rŠ   r‹   )r˜   r–   r—   rŠ   r‹   rŒ   r�   r™   s          €r7   r•   zPyPDFium2Parser.__init__Ñ  sa   ø€ ôL 	‰ÑÔØÐ)Ñ)ÜÐ:Ó;Ð;Ø,ˆÔÙ¡-Ü.Ó0ˆMØ*ˆÔØ#6ˆÔ Ø ˆŒØˆŒ	Ø.ˆÕrJ   c              #  óT  K  — 	 ddl }t        j                  5  |j	                  «       5 }d}	  |j
                  || j                  d¬«      }g }ddddœt        |j                  «       «      z  }|j                  |d	<   t        |«      |d
<   t        |«      D ]ã  \  }}|j                  «       }	dj                  |	j                  «       j                  «       «      }
|	j!                  «        | j#                  |«      }t%        |g|
«      j'                  «       }|j!                  «        | j(                  dk(  r5|j+                  d«      s|dz  }t-        |t/        i |¥d|i¥«      ¬«      –— ŒÓ|j1                  |«       Œå | j(                  dk(  r1t-        | j2                  j                  |«      t/        |«      ¬«      –— |r|j!                  «        	 ddd«       ddd«       y# t        $ r t        d«      ‚w xY w# |r|j!                  «        w w xY w# 1 sw Y   ŒBxY w# 1 sw Y   yxY w­w)r›   r   NzKpypdfium2 package not found, please install it with `pip install pypdfium2`T)r—   Ú	autocloseÚ	PyPDFium2r+   r¤   r9   r<   r-   rL   r¦   r’   )Ú	pypdfium2r/   r±  r‡  r¨   ÚPdfDocumentr—   ri   Úget_metadata_dictr9   r«   r­   Úget_textpager0   Úget_text_rangeÚ
splitlinesÚcloser�  r†   rd   rŠ   rR  r   rU   r°   r‹   )r˜   rF   r¶  rX   r²   rŽ  r³   rµ   rL   Ú	text_pagerx   Úimage_from_pager}   s                r7   r·   zPyPDFium2Parser.lazy_parse  s-  è ø€ ð	Ûô ×"Ó"Ø×!Ñ!Ô# yØ!�
ð1+Ø!6 ×!6Ñ!6Ø!¨D¯M©MÀTô"�Jð $&�Lð %0Ø#.Ø(*ñ$ô (¨
×(DÑ(DÓ(FÓGñ	$H�Lð
 .2¯[©[�L Ñ*Ü25°j³/�L Ñ/ä-6°zÖ-BÑ)˜ TØ$(×$5Ñ$5Ó$7˜	Ø)-¯©Ø%×4Ñ4Ó6×AÑAÓCó*˜ð "Ÿ™Ô)Ø*.×*HÑ*HÈÓ*N˜Ü#9Ø,Ð-¨~ó$ç™%›'ð !ð Ÿ
™
œàŸ9™9¨Ò.à#+×#4Ñ#4°TÔ#:Ø (¨DÑ 0 Ü"*Ø-5Ü);ð%&Ø*6ð%&à(.°ñ%&ó*"ô#ó ð )×/Ñ/°Õ9ð5 .Cð8 —y‘y HÒ,Ü&Ø)-×)=Ñ)=×)BÑ)BÀ<Ó)PÜ%7¸Ó%Eôò ñ
 "Ø"×(Ñ(Õ*÷g $÷ #Ð"øô ò 	Üð+óð ð	ûñv "Ø"×(Ñ(Õ*ð "ú÷e $Ð#ú÷ #Ð"üsa   ‚H(„G  ˆH(˜H©H­FG8Æ<HÇHÇ	H(Ç G5Ç5H(Ç8HÈHÈH	ÈHÈH%È!H(c                óÜ  — | j                   syddlm} t        |j	                  |j
                  f¬«      «      }|syg }|D �]   }t        j                  «       }|j                  «       j                  «       }|j                  dk  rŒFt        j                  ||j                  «       j                  «       «       t        j                  |j                  «       d¬«      }t!        | j                   j#                  |«      «      j$                  }	|j'                  t)        ||	| j*                  «      «       |j-                  «        �Œ t.        j1                  t2        j5                  |«      ¬«      S )	r¹   r+   r   N)rv   rv  r—  rÀ   rÂ   )rŒ   Úpypdfium2.rawÚrawr  Úget_objectsÚFPDF_PAGEOBJ_IMAGErÒ   rÓ   Ú
get_bitmapÚto_numpyÚsizer›  rÙ   r   rÚ   rÛ   rÜ   r·   r§   r°   rI   r�   r¼  rÝ   rH   rÞ   r0   )
r˜   rL   Úpdfium_cr1   Ú
str_imagesrŸ  rå   rá   rF   Útext_from_images
             r7   r�  z)PyPDFium2Parser._extract_images_from_pageR  s!  € ð ×!Ò!Øå(ä�d×&Ñ&¨x×/JÑ/JÐ.LÐ&ÓMÓNˆÙØØˆ
ÜˆEÜŸ*™*›,ˆKØ×'Ñ'Ó)×2Ñ2Ó4ˆHØ�}‰}˜qÒ ØÜ�J‰J�{ E×$4Ñ$4Ó$6×$?Ñ$?Ó$AÔBÜ—>‘> +×"6Ñ"6Ó"8ÐDWÔXˆDÜ" 4×#5Ñ#5×#@Ñ#@ÀÓ#FÓG×TÑTˆOØ×ÑÜ# D¨/¸4×;SÑ;SÓTôð �K‰KŽMð ô !×'Ñ'´<×3DÑ3DÀZÓ3PÐ'ÓQÐQrJ   rW  )r–   rƒ   r—   r…   rŠ   rç   r‹   r_   rŒ   rè   r�   ré   r„   rC  rë   )rL   zpypdfium2._helpers.page.PdfPager„   r_   )rí   rî   rï   rð   r®  r¯  r‡  rñ   r•   r·   r�  rò   ró   s   @r7   r±  r±  š  sŒ   ø„ ñ0ðh ˆI�N‰NÓ€Eð  %ð0/ð #'Ø*0Ø7Ø7;ØKQñ0/àð0/ð  ð	0/ð
 (ð0/ð ð0/ð 5ð0/ð Ið0/ð 
õ0/ódM+÷^RrJ   r±  c                  óF   — e Zd ZdZ	 	 	 d	 	 	 	 	 	 	 dd„Zd	d„Zd
d„Zd
d„Zy)ÚPDFPlumberParserzParse `PDF` with `PDFPlumber`.Nc                óp   — 	 ddl }|xs i | _        || _        || _        y# t        $ r t        d«      ‚w xY w)zØInitialize the parser.

        Args:
            text_kwargs: Keyword arguments to pass to ``pdfplumber.Page.extract_text()``
            dedupe: Avoiding the error of duplicate characters if `dedupe=True`.
        r   NzEpillow package not found, please install it with `pip install pillow`)rÄ   r/   ra  Údeduper–   )r˜   ra  rÍ  r–   rÄ   s        r7   r•   zPDFPlumberParser.__init__v  sI   € ð	Ûð
 'Ò,¨"ˆÔØˆŒØ,ˆÕøô ò 	ÜØWóð ð	ús   ‚   5c              #  ó\  K  — ddl }|j                  «       5 } |j                  |«      }|j                  D ��cg c]À  }t	        | j                  |«      dz   | j                  |«      z   t        |j                  |j                  |j                  dz
  t        |j                  «      dœfi |j                  D �ci c]6  }t        |j                  |   «      t        t        fv r||j                  |   “Œ8 c}¤Ž¬«      ‘ŒÂ c}}E d{  –—†  ddd«       yc c}w c c}}w 7 Œ# 1 sw Y   yxY w­w)úLazily parse the blob.r   Nr-   r,   )r9   rX   rL   r<   r¦   )Ú
pdfplumberr¨   rÑ   r¬   r   Ú_process_page_contentr�  rª   r9   rµ   r«   rT   r^   r_   rS   )r˜   rF   rÐ  rX   r  rL   rg   s          r7   r·   zPDFPlumberParser.lazy_parseŒ  s&  è ø€ ãà×ÑÔ 9Ø!�*—/‘/ )Ó,ˆCð*  ŸIšIô'ñ& &�Dô% Ø!%×!;Ñ!;¸DÓ!AØñ"à×4Ñ4°TÓ:ñ";ô "à&*§k¡kØ)-¯©Ø$(×$4Ñ$4°qÑ$8Ü+.¨s¯y©y«>ñ	ñð &)§\¢\óá%1 Ü# C§L¡L°¡OÓ4¼¼c¸
ÑBð ˜sŸ|™|¨A™Ñ.Ø%1ññö	ð$ &ò'÷ ð ÷  Ðùòùóð ø÷  ÐüsL   ‚D,—"D ¹A>DÂ7;DÃ2DÃ>D ÄDÄD Ä
	D,ÄDÄD Ä D)Ä%D,c                ó¦   — | j                   r* |j                  «       j                  di | j                  ¤ŽS  |j                  di | j                  ¤ŽS )z)Process the page content based on dedupe.rp   )rÍ  Údedupe_charsrŸ   ra  )r˜   rL   s     r7   rÑ  z&PDFPlumberParser._process_page_content©  sJ   € à�;Š;Ø3�4×$Ñ$Ó&×3Ñ3ÑG°d×6FÑ6FÑGÐGØ ˆt× Ñ Ñ4 4×#3Ñ#3Ñ4Ð4rJ   c                óÞ  — ddl m} | j                  syg }|j                  D �]>  }|d   d   j                  t
        v rÒ|d   d   dk(  rd|j                  t        j                  |j                  d|d   d	   |d   d
   f|d   j                  «       «      j                  d«      «      «       Œ‹|j                  t        j                  |d   j                  «       t        j                  ¬«      j                  |d   d
   |d   d	   d«      «       Œî|d   d   j                  t        v r$|j                  |d   j                  «       «       �Œ*t!        j"                  d«       �ŒA t%        |«      S )z8Extract images from page and get the text with RapidOCR.r   rº   r+   r‚  ÚFilterÚBitsPerComponentr,   Ú1ÚWidthÚHeightÚLr¼   rm   r¾   )rÄ   r»   r–   r1   r  rÉ   r°   rÊ   rÐ   Ú	frombytesrÌ   ÚconvertrË   rÍ   rÎ   rÏ   ÚwarningsÚwarnr8   )r˜   rL   r»   r1   r4   s        r7   r�  z*PDFPlumberParser._extract_images_from_page¯  sH  € åà×"Ò"ØàˆØ—;•;ˆCØ�8‰}˜XÑ&×+Ñ+Ô/GÑGØ�x‘=Ð!3Ñ4¸Ò9Ø—M‘MÜŸ™Ø!ŸO™OØ #Ø!$ X¡¨wÑ!7¸¸X¹ÀxÑ9PÐ QØ # H¡× 6Ñ 6Ó 8ó÷ &™g c›lóõð —M‘MÜŸ™ c¨(¡m×&<Ñ&<Ó&>ÄbÇhÁhÔO×WÑWØ ™M¨(Ñ3°S¸±]À7Ñ5KÈRóõð
 �X‘˜xÑ(×-Ñ-Ô1FÑFØ—‘˜c (™m×4Ñ4Ó6Ö7ä—‘Ð3Ö4ð+ ô. 1°Ó8Ð8rJ   )NFF)ra  zOptional[Mapping[str, Any]]rÍ  rƒ   r–   rƒ   r„   rC  rë   )rL   zpdfplumber.page.Pager„   r_   )rí   rî   rï   rð   r•   r·   rÑ  r�  rp   rJ   r7   rË  rË  s  sJ   „ Ù(ð 48ØØ$ð	-à0ð-ð ð-ð ð	-ð
 
ó-ó,ó:5ô9rJ   rË  c                  ó:   — e Zd ZdZ	 	 dddœ	 	 	 	 	 	 	 dd„Zdd„Zy)	ÚAmazonTextractPDFParsera—  Send `PDF` files to `Amazon Textract` and parse them.

    For parsing multi-page PDFs, they have to reside on S3.

    The AmazonTextractPDFLoader calls the
    [Amazon Textract Service](https://aws.amazon.com/textract/)
    to convert PDFs into a Document structure.
    Single and multi-page documents are supported with up to 3000 pages
    and 512 MB of size.

    For the call to be successful an AWS account is required,
    similar to the
    [AWS CLI](https://docs.aws.amazon.com/cli/latest/userguide/cli-chap-configure.html)
    requirements.

    Besides the AWS configuration, it is very similar to the other PDF
    loaders, while also supporting JPEG, PNG and TIFF and non-native
    PDF formats.

    ```python
    from langchain_community.document_loaders import AmazonTextractPDFLoader
    loader=AmazonTextractPDFLoader("example_data/alejandro_rosalez_sample-small.jpeg")
    documents = loader.load()
    ```

    One feature is the linearization of the output.
    When using the features LAYOUT, FORMS or TABLES together with Textract

    ```python
    from langchain_community.document_loaders import AmazonTextractPDFLoader
    # you can mix and match each of the features
    loader=AmazonTextractPDFLoader(
        "example_data/alejandro_rosalez_sample-small.jpeg",
        textract_features=["TABLES", "LAYOUT"])
    documents = loader.load()
    ```

    it will generate output that formats the text in reading order and
    try to output the information in a tabular structure or
    output the key/value pairs with a colon (key: value).
    This helps most LLMs to achieve better accuracy when
    processing these texts.

    ``Document`` objects are returned with metadata that includes the ``source`` and
    a 1-based index of the page number in ``page``. Note that ``page`` represents
    the index of the result returned from Textract, not necessarily the as-written
    page number in the document.

    N)Úlinearization_configc               ó¨  — 	 ddl }ddlmc m} || _        || _        |�%|D �cg c]  }|j                  |«      ‘Œ c}| _        ng | _        |�|| _        n$| j
                  j                  dddd¬«      | _        |s	 ddl}|j                  d	«      | _        y|| _        yc c}w # t        $ r t        d«      ‚w xY w# t        $ r t        d
«      ‚w xY w)a5  Initializes the parser.

        Args:
            textract_features: Features to be used for extraction, each feature
                               should be passed as an int that conforms to the enum
                               `Textract_Features`, see `amazon-textract-caller` pkg
            client: boto3 textract client
            linearization_config: Config to be used for linearization of the output
                                  should be an instance of TextLinearizationConfig from
                                  the `textractor` pkg
        r   NTz# z## Ú*)Úhide_figure_layoutÚtitle_prefixÚsection_header_prefixÚlist_element_prefixz½Could not import amazon-textract-caller or amazon-textract-textractor python package. Please install it with `pip install amazon-textract-caller` & `pip install amazon-textract-textractor`.ÚtextractzRCould not import boto3 python package. Please install it with `pip install boto3`.)ÚtextractcallerÚtextractor.entities.documentÚentitiesÚdocumentÚtcÚ
textractorÚTextract_FeaturesÚtextract_featuresrá  r   r/   Úboto3ÚclientÚboto3_textract_client)r˜   rð  rò  rá  rí  rî  Úfrñ  s           r7   r•   z AmazonTextractPDFParser.__init__  s  € ð&	Û'ß=Ð=àˆDŒGØ(ˆDŒOà Ð,á5Fó*Ù5F°�B×(Ñ(¨Õ+Ð5Fñ*�Õ&ð *,�Ô&à#Ð/Ø,@�Õ)à,0¯O©O×,SÑ,SØ'+Ø!%Ø*/Ø(+ð	 -Tó -�Ô)ñ ðÛà-2¯\©\¸*Ó-E�Õ*ð *0ˆDÕ&ùòE*øô ò 	Üð<óð ð	ûô ò Ü!ðBóð ðús'   ‚!B$ £B»>B$ Á<B< ÂB$ Â$B9Â<Cc              #  óÞ  K  — |j                   rt        t        |j                   «      «      nd}|ra|j                  dk(  rR|j                  rF| j
                  j                  t        |j                   «      | j                  | j                  ¬«      }n_| j
                  j                  |j                  «       | j                  | j
                  j                  j                  | j                  ¬«      }| j                  j                  j                  |«      }t        |j                   «      D ]>  \  }}t        |j#                  | j$                  ¬«      |j&                  |dz   dœ¬«      –— Œ@ y­w)	zùIterates over the Blob pages and returns an Iterator with a Document
        for each page, like the other parsers If multi-page document, blob.path
        has to be set to the S3 URI and for single page docs
        the blob.data is taken
        NÚs3)Úinput_documentÚfeaturesró  )r÷  rø  Ú	call_moderó  )Úconfigr,   ©r9   rL   r¦   )Úpathr   r_   ÚschemeÚnetlocrí  Úcall_textractrð  ró  Úas_bytesÚTextract_Call_ModeÚ
FORCE_SYNCrî  r   rÑ   r­   r¬   r8  rá  r9   )r˜   rF   Úurl_parse_resultÚtextract_response_jsonrì  ÚidxrL   s          r7   r·   z"AmazonTextractPDFParser.lazy_parseC  s%  è ø€ ð 8<·y²yœ8¤C¨¯	©	£NÔ3ÀdÐñ Ø ×'Ñ'¨4Ò/Ø ×'Ò'à%)§W¡W×%:Ñ%:Ü" 4§9¡9›~Ø×/Ñ/Ø&*×&@Ñ&@ð &;ó &Ñ"ð &*§W¡W×%:Ñ%:Ø#Ÿ}™}›Ø×/Ñ/ØŸ'™'×4Ñ4×?Ñ?Ø&*×&@Ñ&@ð	 &;ó &Ð"ð —?‘?×+Ñ+×0Ñ0Ð1GÓHˆä" 8§>¡>Ö2‰IˆC�ÜØ!Ÿ]™]°$×2KÑ2K˜]ÓLØ$(§K¡K¸¸q¹ÑAôó ñ 3ùs   ‚E+E-)NN)rð  zOptional[Sequence[int]]rò  zOptional[Any]rá  z!Optional[TextLinearizationConfig]r„   rC  rë   )rí   rî   rï   rð   r•   r·   rp   rJ   r7   rà  rà  Ñ  sN   „ ñ0ðh 6:Ø $ð=0ð
 CGñ=0à2ð=0ð ð=0ð
 @ð=0ð 
ó=0ô~!rJ   rà  c                  ó(   — e Zd ZdZdd„Zdd„Zdd„Zy)	ÚDocumentIntelligenceParserzjLoads a PDF with Azure Document Intelligence
    (formerly Form Recognizer) and chunks at character level.c                óJ   — t        j                  d«       || _        || _        y )Na<  langchain_community.document_loaders.parsers.pdf.DocumentIntelligenceParserand langchain_community.document_loaders.pdf.DocumentIntelligenceLoader are deprecated. Please upgrade to langchain_community.document_loaders.DocumentIntelligenceLoader for any file parsing purpose using Azure Document Intelligence service.)rÝ  rÞ  rò  Úmodel)r˜   rò  r	  s      r7   r•   z#DocumentIntelligenceParser.__init__k  s#   € Ü�‰ðô	
ð ˆŒØˆ�
rJ   c              #  óî   K  — |j                   D ]]  }dj                  |j                  D �cg c]  }|j                  ‘Œ c}«      }t	        ||j
                  |j                  dœ¬«      }|–— Œ_ y c c}w ­w)NÚ rû  r¦   )r¬   r0   rk  rG   r   r9   rµ   )r˜   rF   r5   ÚpÚlinerG   Úds          r7   Ú_generate_docsz)DocumentIntelligenceParser._generate_docsw  sc   è ø€ Ø—”ˆAØ—h‘h¸¿ºÓA¹° §£¸ÑAÓBˆGäØ$à"Ÿk™kØŸM™MñôˆAð ‹Gñ ùÚAùs   ‚)A5«A0
¾7A5c              #  óþ   K  — |j                  «       5 }| j                  j                  | j                  |«      }|j	                  «       }| j                  ||«      }|E d{  –—†  ddd«       y7 Œ# 1 sw Y   yxY w­w)rÏ  N)r¨   rò  Úbegin_analyze_documentr	  r5   r  )r˜   rF   Úfile_objÚpollerr5   Údocss         r7   r·   z%DocumentIntelligenceParser.lazy_parse„  sj   è ø€ ð ×ÑÔ 8Ø—[‘[×7Ñ7¸¿
¹
ÀHÓMˆFØ—]‘]“_ˆFà×&Ñ& t¨VÓ4ˆDà�OˆO÷  Ðð ø÷  Ðüs/   ‚A=“AA1Á!A/Á"A1Á&	A=Á/A1Á1A:Á6A=N)rò  r   r	  r_   )rF   r   r5   r   r„   rì   rë   )rí   rî   rï   rð   r•   r  r·   rp   rJ   r7   r  r  g  s   „ ñAó
óô	rJ   r  )r1   z,Sequence[Union[Iterable[np.ndarray], bytes]]r„   r_   )rF   r   rG   r_   rH   r_   r„   r_   )rT   rX  r„   rX  )rw   r‚   rx   r_   r„   r_   )Crð   Ú
__future__r   rD   rÒ   Úloggingr®  rÝ  r   Úpathlibr   Útempfiler   Útypingr   r   r	   r
   r   r   r   r   r   r   r   Úurllib.parser   r›  rÊ   Úlangchain_core.documentsr   Ú)langchain_community.document_loaders.baser   Ú1langchain_community.document_loaders.blob_loadersr   Ú3langchain_community.document_loaders.parsers.imagesr   r   rÐ  r…  r    r¶  Ú)textractor.data.text_linearization_configr   rÏ   rÉ   r8   Ú	getLoggerrí   rÔ   rÝ   rÞ   r¦  rñ   rM   rI   rU   ri   rt   r†   rˆ   rõ   r[  r±  rË  rà  r  rp   rJ   r7   Ú<module>r!     sE  ðÙ .å "ã Û 	Û Û Û Ý Ý Ý '÷÷ ÷ ñ õ "ã Û Ý -å DÝ B÷ñ
 ÛÛÛÛÝQâ9Ð òÐ ð"Ø8ðàóð> 
ˆ×	Ñ	˜8Ó	$€à*Ð Ø€Ø€Ø!Ð âUÐ óó,ó&#ðN Ø
ðÐ ó2ôje
�.ô e
ôPJ�^ô JôZ
o�Nô oôdVR�nô VRôr[9�~ô [9ô|S˜nô Sôl& õ &rJ   