§
    ‚Štj�@  ã                   ó<  — d dl mZmZ ddlmZ ddlmZmZ ddlm	Z	m
Z
mZmZ ddlmZmZmZ ddlmZmZ  e¦   «         rd dlZ G d	„ d
e
d¬¦  «        ZdZd„  ed¦  «        D ¦   «         d„  ed¦  «        D ¦   «         z   Zd„ Ze G d„ de¦  «        ¦   «         ZdgZdS )é    )ÚOptionalÚUnioné   )ÚBatchFeature)Ú
ImageInputÚmake_flat_list_of_images)ÚMultiModalDataÚProcessingKwargsÚProcessorMixinÚUnpack)Ú
AddedTokenÚPreTokenizedInputÚ	TextInput)Úauto_docstringÚis_torch_availableNc                   ó(   — e Zd ZddidddœddidœZd	S )
ÚColPaliProcessorKwargsÚpaddingÚlongestÚchannels_firstT)Údata_formatÚdo_convert_rgbÚreturn_tensorsÚpt)Útext_kwargsÚimages_kwargsÚcommon_kwargsN)Ú__name__Ú
__module__Ú__qualname__Ú	_defaults© ó    úl/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/colpali/processing_colpali.pyr   r   #   sB   € € € € € ð �yð
ð ,Ø"ð
ð 
ð +¨DÐ1ð	ð 	€I€I€Ir#   r   F)Útotalz<image>c                 ó   — g | ]	}d |d›d�‘Œ
S )z<locz0>4ú>r"   ©Ú.0Úis     r$   ú
<listcomp>r+   1   s"   € Ð5Ð5Ð5 A��q����Ð5Ð5Ð5r#   i   c                 ó   — g | ]	}d |d›d�‘Œ
S )z<segz0>3r'   r"   r(   s     r$   r+   r+   1   s"   € Ð8]Ð8]Ð8]ÈQ¸À¸¸¸¸Ð8]Ð8]Ð8]r#   é€   c                 ó    — ||z  |z  › |› | › d�S )aZ  
    Builds a string from the input prompt and image tokens.
    For example, for the call:
    build_string_from_input(
        prompt="Prefix str"
        bos_token="<s>",
        image_seq_len=3,
        image_token="<im>",
    )
    The output will be:
    "<im><im><im><s>Initial str"
    Args:
        prompt (`list[Union[str, ImageInput]]`): The input prompt.
        bos_token (`str`): The beginning of sentence token.
        image_seq_len (`int`): The length of the image sequence.
        image_token (`str`): The image token.
        num_images (`int`): Number of images in the prompt.
    ú
r"   ©ÚpromptÚ	bos_tokenÚimage_seq_lenÚimage_tokenÚ
num_imagess        r$   Úbuild_string_from_inputr6   4   s'   € ð& ˜MÑ)¨JÑ6ÐM¸	ÐMÀ6ÐMÐMÐMÐMr#   c                   ó°  ‡ — e Zd Z	 	 	 	 	 ddedefˆ fd„Ze	 	 ddedz  deez  e	e         z  e	e         z  d	e
e         d
efd„¦   «         Zdd„Zed„ ¦   «         Zed
efd„¦   «         Z	 ddedz  d	e
e         d
efd„Zdee	e         z  d	e
e         d
efd„Z	 	 	 ddede	d         f         dede	d         f         deded         dedef         d
dfd„Zˆ xZS ) ÚColPaliProcessorNúDescribe the image.ú
Question: Úvisual_prompt_prefixÚquery_prefixc                 ó  •— || _         || _        t          |d¦  «        st          d¦  «        ‚|j        | _        t          |d¦  «        s]t          t          dd¬¦  «        }d|gi}|                     |¦  «         |                     t          ¦  «        | _	        t          | _
        n|j	        | _	        |j
        | _
        |                     t          ¦  «         d|_        d|_        t          ¦   «                              |||¬¦  «         d	S )
a!  
        visual_prompt_prefix (`str`, *optional*, defaults to `"Describe the image."`):
            A string that gets tokenized and prepended to the image tokens.
        query_prefix (`str`, *optional*, defaults to `"Question: "`):
            A prefix to be used for the query.
        Úimage_seq_lengthz;Image processor is missing an `image_seq_length` attribute.r4   FT)Ú
normalizedÚspecialÚadditional_special_tokens)Úchat_templateN)r;   r<   ÚhasattrÚ
ValueErrorr>   r   ÚIMAGE_TOKENÚadd_special_tokensÚconvert_tokens_to_idsÚimage_token_idr4   Ú
add_tokensÚEXTRA_TOKENSÚadd_bos_tokenÚadd_eos_tokenÚsuperÚ__init__)	ÚselfÚimage_processorÚ	tokenizerrB   r;   r<   r4   Útokens_to_addÚ	__class__s	           €r$   rN   zColPaliProcessor.__init__L   s	  ø€ ð %9ˆÔ!Ø(ˆÔÝ�Ð(:Ñ;Ô;ð 	\ÝÐZÑ[Ô[Ð[à /Ô @ˆÔå�y -Ñ0Ô0ð 	5Ý$¥[¸UÈDÐQÑQÔQˆKØ8¸;¸-ÐHˆMØ×(Ò(¨Ñ7Ô7Ð7Ø"+×"AÒ"AÅ+Ñ"NÔ"NˆDÔÝ*ˆDÔÐà"+Ô":ˆDÔØ(Ô4ˆDÔà×Ò�\Ñ*Ô*Ð*Ø"'ˆ	ÔØ"'ˆ	Ôå‰Œ×Ò˜¨)À=ÐÑQÔQÐQÐQÐQr#   ÚimagesÚtextÚkwargsÚreturnc                 óÞ  ‡ —  ‰ j         t          fd‰ j        j        i|¤Ž}|d                              dd¦  «        }d}|€|€t          d¦  «        ‚|�|�t          d¦  «        ‚|��)‰ j                             |¦  «        }t          |¦  «        }‰ j	        gt          |¦  «        z  }ˆ fd„|D ¦   «         }ˆ fd	„t          ||¦  «        D ¦   «         } ‰ j        |fi |d
         ¤Žd         }	|d                              dd¦  «        �|d         dxx         ‰ j        z  cc<    ‰ j        |fd|i|d         ¤Ž}
i |
¥d|	i¥}|r=|
d                              |
d         dk    d¦  «        }|                     d|i¦  «         t!          |¬¦  «        S |�Út#          |t$          ¦  «        r|g}n?t#          |t&          ¦  «        rt#          |d         t$          ¦  «        st          d¦  «        ‚|€
‰ j        dz  }g }|D ]4}‰ j        j        ‰ j        z   |z   |z   dz   }|                     |¦  «         Œ5|d                              dd¦  «        |d         d<    ‰ j        |fd|i|d         ¤Ž}|S dS )a  
        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
        Útokenizer_init_kwargsr   ÚsuffixNTz&Either text or images must be providedz5Only one of text or images can be processed at a timec                 óD   •— g | ]}‰j                              |¦  «        ‘ŒS r"   )rP   Úprocess_image)r)   ÚimagerO   s     €r$   r+   z-ColPaliProcessor.__call__.<locals>.<listcomp>”   s*   ø€ ÐTÐTÐTÀE�dÔ*×8Ò8¸Ñ?Ô?ÐTÐTÐTr#   c                 ó®   •— g | ]Q\  }}t          |‰j        j        ‰j        t          t          |t          ¦  «        rt          |¦  «        nd ¬¦  «        ‘ŒRS )é   r0   )r6   rQ   r2   r>   rE   Ú
isinstanceÚlistÚlen)r)   r1   Ú
image_listrO   s      €r$   r+   z-ColPaliProcessor.__call__.<locals>.<listcomp>–   so   ø€ ð 	ð 	ð 	ñ '�F˜Jõ (Ø!Ø"œnÔ6Ø"&Ô"7Ý +Ý2<¸ZÍÑ2NÔ2NÐU�s :™œ˜ÐTUðñ ô ð	ð 	ð 	r#   r   Úpixel_valuesÚ
max_lengthÚreturn_token_type_idsÚ	input_idsÚtoken_type_idsr   iœÿÿÿÚlabels)Údataz*Text must be a string or a list of stringsé
   r/   é2   )Ú_merge_kwargsr   rQ   Úinit_kwargsÚpoprD   rP   Úfetch_imagesr   r;   rb   ÚzipÚgetr>   Úmasked_fillÚupdater   r`   Ústrra   Úquery_augmentation_tokenr2   r<   Úappend)rO   rT   rU   rV   Úoutput_kwargsrZ   rf   Ú	texts_docÚinput_stringsrd   ÚinputsÚreturn_datari   Útexts_queryÚqueryÚbatch_querys   `               r$   Ú__call__zColPaliProcessor.__call__q   sC  ø€ ð" +˜Ô*Ý"ð
ð 
à"&¤.Ô"<ð
ð ð
ð 
ˆð
 ˜}Ô-×1Ò1°(¸DÑAÔAˆà $Ðàˆ<˜F˜NÝÐEÑFÔFÐFØÐ Ð 2ÝÐTÑUÔUÐUàÑØÔ)×6Ò6°vÑ>Ô>ˆFÝ-¨fÑ5Ô5ˆFØÔ2Ð3µc¸&±k´kÑAˆIØTÐTÐTÐTÈVÐTÑTÔTˆFð	ð 	ð 	ð 	õ +.¨i¸Ñ*@Ô*@ð	ñ 	ô 	ˆMð 0˜4Ô/°ÐYÐY¸-ÈÔ:XÐYÐYÐZhÔiˆLð ˜]Ô+×/Ò/°¸dÑCÔCÐOØ˜mÔ,¨\Ð:Ð:Ô:¸dÔ>SÑSÐ:Ð:Ñ:à#�T”^Øðð à&;ðð   Ô.ðð ˆFð C˜VÐB ^°\ÐBÐBˆKà$ð 7Ø Ô,×8Ò8¸Ð@PÔ9QÐUVÒ9VÐX\Ñ]Ô]�Ø×"Ò" H¨fÐ#5Ñ6Ô6Ð6å [Ð1Ñ1Ô1Ð1àÐÝ˜$¥Ñ$Ô$ð OØ�v��Ý  ¥tÑ,Ô,ð Oµ¸DÀ¼GÅSÑ1IÔ1Ið OÝ Ð!MÑNÔNÐNàˆ~ØÔ6¸Ñ;�à%'ˆKØð *ð *�ØœÔ0°4Ô3DÑDÀuÑLÈvÑUÐX\Ñ\�Ø×"Ò" 5Ñ)Ô)Ð)Ð)à9FÀ}Ô9U×9YÒ9YÐZfÐhjÑ9kÔ9kˆM˜-Ô(¨Ñ6à(˜$œ.Øðð à&;ðð   Ô.ðð ˆKð Ðð- Ðr#   c                 ó¨   — i }|�C| j         gt          |¦  «        z  }dgt          |¦  «        z  }|                     ||dœ¦  «         t          di |¤ŽS )a¸  
        Computes the number of placeholder tokens needed for multimodal inputs with the given sizes.

        Args:
            image_sizes (list[list[str]], *optional*):
                The input sizes formatted as (height, width) per each image.
        Returns:
            `MultiModalData`: A `MultiModalData` object holding number of tokens per each of the provided
            input modalities, along with other useful data.
        Nr_   )Únum_image_tokensÚnum_image_patchesr"   )r>   rb   rt   r	   )rO   Úimage_sizesrV   Úvision_datar‚   rƒ   s         r$   Ú_get_num_multimodal_tokensz+ColPaliProcessor._get_num_multimodal_tokensÌ   so   € ð ˆØÐ"Ø $Ô 5Ð6½¸[Ñ9IÔ9IÑIÐØ!" ¥c¨+Ñ&6Ô&6Ñ 6ÐØ×ÒÐ4DÐ[lÐmÐmÑnÔnÐnÝÐ,Ð, Ð,Ð,Ð,r#   c                 ó`   — | j         j        ddgz   }| j        j        }t          ||z   ¦  «        S )Nrh   ri   )rQ   Úmodel_input_namesrP   ra   )rO   Útokenizer_input_namesÚimage_processor_input_namess      r$   rˆ   z"ColPaliProcessor.model_input_namesÞ   s:   € à $¤Ô @ÐDTÐV^ÐC_Ñ _ÐØ&*Ô&:Ô&LÐ#ÝÐ)Ð,GÑGÑHÔHÐHr#   c                 ó   — | j         j        S )zŠ
        Return the query augmentation token.

        Query augmentation buffers are used as reasoning buffers during inference.
        )rQ   Ú	pad_token)rO   s    r$   rv   z)ColPaliProcessor.query_augmentation_tokenä   s   € ð Œ~Ô'Ð'r#   c                 ó    —  | j         dd|i|¤ŽS )a  
        Prepare for the model one or several image(s). This method is a wrapper around the `__call__` method of the ColPaliProcessor's
        [`ColPaliProcessor.__call__`].

        This method forwards the `images` and `kwargs` arguments to the image processor.

        Args:
            images (`PIL.Image.Image`, `np.ndarray`, `torch.Tensor`, `list[PIL.Image.Image]`, `list[np.ndarray]`, `list[torch.Tensor]`):
                The image or batch of images to be prepared. Each image can be a PIL image, NumPy array or PyTorch
                tensor. In case of a NumPy array/PyTorch tensor, each image should be of shape (C, H, W), where C is a
                number of channels, H and W are image height and width.
            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors of a particular framework. Acceptable values are:

                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return NumPy `np.ndarray` objects.

        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
        rT   r"   ©r€   )rO   rT   rV   s      r$   Úprocess_imageszColPaliProcessor.process_imagesí   s!   € ð> ˆtŒ}Ð5Ð5 FÐ5¨fÐ5Ð5Ð5r#   c                 ó    —  | j         dd|i|¤ŽS )ag  
        Prepare for the model one or several texts. This method is a wrapper around the `__call__` method of the ColPaliProcessor's
        [`ColPaliProcessor.__call__`].

        This method forwards the `text` and `kwargs` arguments to the tokenizer.

        Args:
            text (`str`, `list[str]`, `list[list[str]]`):
                The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings
                (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set
                `is_split_into_words=True` (to lift the ambiguity with a batch of sequences).
            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors of a particular framework. Acceptable values are:

                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return NumPy `np.ndarray` objects.

        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
        rU   r"   rŽ   )rO   rU   rV   s      r$   Úprocess_queriesz ColPaliProcessor.process_queries  s!   € ð< ˆtŒ}Ð1Ð1 $Ð1¨&Ð1Ð1Ð1r#   r-   ÚcpuÚquery_embeddingsztorch.TensorÚpassage_embeddingsÚ
batch_sizeÚoutput_dtypeztorch.dtypeÚoutput_deviceztorch.devicec           	      ó8  — t          |¦  «        dk    rt          d¦  «        ‚t          |¦  «        dk    rt          d¦  «        ‚|d         j        |d         j        k    rt          d¦  «        ‚|d         j        |d         j        k    rt          d¦  «        ‚|€|d         j        }g }t	          dt          |¦  «        |¦  «        D �]:}g }t
          j        j        j         	                    ||||z   …         dd¬¦  «        }	t	          dt          |¦  «        |¦  «        D ]�}
t
          j        j        j         	                    ||
|
|z   …         dd¬¦  «        }| 
                    t          j        d	|	|¦  «                             d
¬¦  «        d                              d¬¦  «        ¦  «         Œ‘| 
                    t          j        |d¬¦  «                             |¦  «                             |¦  «        ¦  «         �Œ<t          j        |d¬¦  «        S )aZ  
        Compute the late-interaction/MaxSim score (ColBERT-like) for the given multi-vector
        query embeddings (`qs`) and passage embeddings (`ps`). For ColPali, a passage is the
        image of a document page.

        Because the embedding tensors are multi-vector and can thus have different shapes, they
        should be fed as:
        (1) a list of tensors, where the i-th tensor is of shape (sequence_length_i, embedding_dim)
        (2) a single tensor of shape (n_passages, max_sequence_length, embedding_dim) -> usually
            obtained by padding the list of tensors.

        Args:
            query_embeddings (`Union[torch.Tensor, list[torch.Tensor]`): Query embeddings.
            passage_embeddings (`Union[torch.Tensor, list[torch.Tensor]`): Passage embeddings.
            batch_size (`int`, *optional*, defaults to 128): Batch size for computing scores.
            output_dtype (`torch.dtype`, *optional*, defaults to `torch.float32`): The dtype of the output tensor.
                If `None`, the dtype of the input embeddings is used.
            output_device (`torch.device` or `str`, *optional*, defaults to "cpu"): The device of the output tensor.

        Returns:
            `torch.Tensor`: A tensor of shape `(n_queries, n_passages)` containing the scores. The score
            tensor is saved on the "cpu" device.
        r   zNo queries providedzNo passages providedz/Queries and passages must be on the same devicez-Queries and passages must have the same dtypeNT)Úbatch_firstÚpadding_valuezbnd,csd->bcnsr   )Údimé   r_   )rb   rD   ÚdeviceÚdtypeÚrangeÚtorchÚnnÚutilsÚrnnÚpad_sequencerw   ÚeinsumÚmaxÚsumÚcatÚto)rO   r“   r”   r•   r–   r—   Úscoresr*   Úbatch_scoresÚbatch_queriesÚjÚbatch_passagess               r$   Úscore_retrievalz ColPaliProcessor.score_retrieval.  s*  € õ@ ÐÑ Ô  AÒ%Ð%ÝÐ2Ñ3Ô3Ð3ÝÐ!Ñ"Ô" aÒ'Ð'ÝÐ3Ñ4Ô4Ð4à˜AÔÔ%Ð);¸AÔ)>Ô)EÒEÐEÝÐNÑOÔOÐOà˜AÔÔ$Ð(:¸1Ô(=Ô(CÒCÐCÝÐLÑMÔMÐMàÐØ+¨AÔ.Ô4ˆLà%'ˆå�q�#Ð.Ñ/Ô/°Ñ<Ô<ð 	]ñ 	]ˆAØ/1ˆLÝ!œHœNÔ.×;Ò;Ø   Q¨¡^Ð!3Ô4À$ÐVWð <ñ ô ˆMõ ˜1�cÐ"4Ñ5Ô5°zÑBÔBð ð �Ý!&¤¤Ô!3×!@Ò!@Ø& q¨1¨z©>Ð'9Ô:ÈÐ\]ð "Añ "ô "�ð ×#Ò#Ý”L °-ÀÑPÔP×TÒTÐYZÐTÑ[Ô[Ð\]Ô^×bÒbÐghÐbÑiÔiñô ð ð ð �MŠM�%œ) L°aÐ8Ñ8Ô8×;Ò;¸LÑIÔI×LÒLÈ]Ñ[Ô[Ñ\Ô\Ð\Ñ\åŒy˜ QÐ'Ñ'Ô'Ð'r#   )NNNr9   r:   )NN)N)r-   Nr’   )r   r   r    ru   rN   r   r   r   r   ra   r   r   r   r€   r†   Úpropertyrˆ   rv   r�   r‘   r   Úintr   r¯   Ú__classcell__)rS   s   @r$   r8   r8   J   s]  ø€ € € € € ð ØØØ$9Ø(ð#Rð #Rð
 "ð#Rð ð#Rð #Rð #Rð #Rð #Rð #RðJ ð %)ØZ^ðXð Xà˜TÑ!ðXð Ð+Ñ+¨d°9¬oÑ=ÀÐEVÔ@WÑWðXð Ð/Ô0ð	Xð
 
ðXð Xð Xñ „^ðXðt-ð -ð -ð -ð$ ðIð Iñ „XðIð
 ð(¨#ð (ð (ð (ñ „Xð(ð %)ð6ð 6à˜TÑ!ð6ð Ð/Ô0ð6ð 
ð	6ð 6ð 6ð 6ðB2à˜$˜yœ/Ñ)ð2ð Ð/Ô0ð2ð 
ð	2ð 2ð 2ð 2ðH Ø04Ø49ð>(ð >(à °°^Ô0DÐ DÔEð>(ð " .°$°~Ô2FÐ"FÔGð>(ð ð	>(ð
 ˜}Ô-ð>(ð ˜^¨SÐ0Ô1ð>(ð 
ð>(ð >(ð >(ð >(ð >(ð >(ð >(ð >(r#   r8   )Útypingr   r   Úfeature_extraction_utilsr   Úimage_utilsr   r   Úprocessing_utilsr	   r
   r   r   Útokenization_utils_baser   r   r   r¢   r   r   r    r   rE   rŸ   rJ   r6   r8   Ú__all__r"   r#   r$   ú<module>r¹      s“  ðð, #Ð "Ð "Ð "Ð "Ð "Ð "Ð "à 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø ?Ð ?Ð ?Ð ?Ð ?Ð ?Ð ?Ð ?Ø XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XØ OÐ OÐ OÐ OÐ OÐ OÐ OÐ OÐ OÐ OØ 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7ð ÐÑÔð Ø€L€L€Lð
ð 
ð 
ð 
ð 
Ð-°Uð 
ñ 
ô 
ð 
ð €Ø5Ð5¨¨¨t©¬Ð5Ñ5Ô5Ð8]Ð8]ÐRWÐRWÐX[ÑR\ÔR\Ð8]Ñ8]Ô8]Ñ]€ðNð Nð Nð, ða(ð a(ð a(ð a(ð a(�~ñ a(ô a(ñ „ða(ðH	 Ð
€€€r#   